{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/ran/2","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":280,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation/papers/ran/1","prev":"/task/code-generation/papers/ran/1","next":"/task/code-generation/papers/ran/3","papers":[{"url":"/paper/citygpt-empowering-urban-spatial-cognition-of","slug":"citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","arxiv_id":"2406.13948","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/citygpt-empowering-urban-spatial-cognition-of#ran","syntology_url":"https://syntology.ai/paper/2406.13948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13948"}},"official":{"repos":["tsinghua-fib-lab/citygpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coderag-bench-can-retrieval-augment-code","slug":"coderag-bench-can-retrieval-augment-code","title":"CodeRAG-Bench: Can Retrieval Augment Code Generation?","date":"2024-06-20","arxiv_id":"2406.14497","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderag-bench-can-retrieval-augment-code#ran","syntology_url":"https://syntology.ai/paper/2406.14497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14497"}},"official":{"repos":["code-rag-bench/code-rag-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-agents-are-state-of-the-art-software","slug":"code-agents-are-state-of-the-art-software","title":"SWT-Bench: Testing and Validating Real-World Bug-Fixes with Code Agents","date":"2024-06-18","arxiv_id":"2406.12952","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/code-agents-are-state-of-the-art-software#ran","syntology_url":"https://syntology.ai/paper/2406.12952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12952"}},"official":{"repos":["logic-star-ai/swt-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/long-code-arena-a-set-of-benchmarks-for-long","slug":"long-code-arena-a-set-of-benchmarks-for-long","title":"Long Code Arena: a Set of Benchmarks for Long-Context Code Models","date":"2024-06-17","arxiv_id":"2406.11612","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/long-code-arena-a-set-of-benchmarks-for-long#ran","syntology_url":"https://syntology.ai/paper/2406.11612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11612"}},"official":{"repos":["jetbrains-research/lca-baselines"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cs-bench-a-comprehensive-benchmark-for-large","slug":"cs-bench-a-comprehensive-benchmark-for-large","title":"CS-Bench: A Comprehensive Benchmark for Large Language Models towards Computer Science Mastery","date":"2024-06-12","arxiv_id":"2406.08587","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cs-bench-a-comprehensive-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.08587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08587"}},"official":{"repos":["csbench/csbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sinklora-enhanced-efficiency-and-chat","slug":"sinklora-enhanced-efficiency-and-chat","title":"SinkLoRA: Enhanced Efficiency and Chat Capabilities for Long-Context Large Language Models","date":"2024-06-09","arxiv_id":"2406.05678","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sinklora-enhanced-efficiency-and-chat#ran","syntology_url":"https://syntology.ai/paper/2406.05678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05678"}},"official":{"repos":["dexter-gt-86/sinklora"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/buffer-of-thoughts-thought-augmented","slug":"buffer-of-thoughts-thought-augmented","title":"Buffer of Thoughts: Thought-Augmented Reasoning with Large Language Models","date":"2024-06-06","arxiv_id":"2406.04271","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":1,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/buffer-of-thoughts-thought-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.04271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04271"}},"official":{"repos":["yangling0818/buffer-of-thought-llm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/semcoder-training-code-language-models-with","slug":"semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","arxiv_id":"2406.01006","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/semcoder-training-code-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2406.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01006"}},"official":{"repos":["arise-lab/semcoder"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deveval-a-manually-annotated-code-generation","slug":"deveval-a-manually-annotated-code-generation","title":"DevEval: A Manually-Annotated Code Generation Benchmark Aligned with Real-World Code Repositories","date":"2024-05-30","arxiv_id":"2405.19856","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deveval-a-manually-annotated-code-generation#ran","syntology_url":"https://syntology.ai/paper/2405.19856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19856"}},"official":{"repos":["seketeam/deveval"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/soap-enhancing-efficiency-of-generated-code","slug":"soap-enhancing-efficiency-of-generated-code","title":"EffiLearner: Enhancing Efficiency of Generated Code via Self-Optimization","date":"2024-05-24","arxiv_id":"2405.15189","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/soap-enhancing-efficiency-of-generated-code#ran","syntology_url":"https://syntology.ai/paper/2405.15189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15189"}},"official":{"repos":["huangd1999/effilearner","huangd1999/soap"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-tuning-strikes-back-customizing","slug":"prompt-tuning-strikes-back-customizing","title":"Prompt Tuning Strikes Back: Customizing Foundation Models with Low-Rank Prompt Adaptation","date":"2024-05-24","arxiv_id":"2405.15282","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompt-tuning-strikes-back-customizing#ran","syntology_url":"https://syntology.ai/paper/2405.15282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15282"}},"official":{"repos":["jabhinav/prompt-tuning-strikes-back-with-lopa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-code-world-models-with-large","slug":"generating-code-world-models-with-large","title":"Generating Code World Models with Large Language Models Guided by Monte Carlo Tree Search","date":"2024-05-24","arxiv_id":"2405.15383","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-code-world-models-with-large#ran","syntology_url":"https://syntology.ai/paper/2405.15383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15383"}},"official":{"repos":["nicoladainese96/code-world-models"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comera-computing-and-memory-efficient","slug":"comera-computing-and-memory-efficient","title":"CoMERA: Computing- and Memory-Efficient Training via Rank-Adaptive Tensor Optimization","date":"2024-05-23","arxiv_id":"2405.14377","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":2,"n_ran_checked":6,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"9 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/comera-computing-and-memory-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.14377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14377"}},"official":{"repos":["ziyangjoy/comera"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/analogcoder-analog-circuit-design-via","slug":"analogcoder-analog-circuit-design-via","title":"AnalogCoder: Analog Circuit Design via Training-Free Code Generation","date":"2024-05-23","arxiv_id":"2405.14918","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analogcoder-analog-circuit-design-via#ran","syntology_url":"https://syntology.ai/paper/2405.14918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14918"}},"official":{"repos":["laiyao1/AnalogCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mapcoder-multi-agent-code-generation-for","slug":"mapcoder-multi-agent-code-generation-for","title":"MapCoder: Multi-Agent Code Generation for Competitive Problem Solving","date":"2024-05-18","arxiv_id":"2405.11403","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mapcoder-multi-agent-code-generation-for#ran","syntology_url":"https://syntology.ai/paper/2405.11403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11403"}},"official":{"repos":["md-ashraful-pramanik/mapcoder"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/granite-code-models-a-family-of-open","slug":"granite-code-models-a-family-of-open","title":"Granite Code Models: A Family of Open Foundation Models for Code Intelligence","date":"2024-05-07","arxiv_id":"2405.04324","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/granite-code-models-a-family-of-open#ran","syntology_url":"https://syntology.ai/paper/2405.04324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04324"}},"official":{"repos":["ibm-granite/granite-code-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/constrained-decoding-for-secure-code","slug":"constrained-decoding-for-secure-code","title":"Constrained Decoding for Secure Code Generation","date":"2024-04-30","arxiv_id":"2405.00218","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/constrained-decoding-for-secure-code#ran","syntology_url":"https://syntology.ai/paper/2405.00218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00218"}},"official":{"repos":["dynamite321/codeguardplus"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/class-level-code-generation-from-natural","slug":"class-level-code-generation-from-natural","title":"Class-Level Code Generation from Natural Language Using Iterative, Tool-Enhanced Reasoning over Repository","date":"2024-04-22","arxiv_id":"2405.01573","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/class-level-code-generation-from-natural#ran","syntology_url":"https://syntology.ai/paper/2405.01573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01573"}},"official":null}},{"url":"/paper/large-language-models-as-test-case-generators","slug":"large-language-models-as-test-case-generators","title":"Large Language Models as Test Case Generators: Performance Evaluation and Enhancement","date":"2024-04-20","arxiv_id":"2404.13340","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-as-test-case-generators#ran","syntology_url":"https://syntology.ai/paper/2404.13340","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13340"}},"official":null}},{"url":"/paper/mmcode-evaluating-multi-modal-code-large","slug":"mmcode-evaluating-multi-modal-code-large","title":"MMCode: Benchmarking Multimodal Large Language Models for Code Generation with Visually Rich Programming Problems","date":"2024-04-15","arxiv_id":"2404.09486","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmcode-evaluating-multi-modal-code-large#ran","syntology_url":"https://syntology.ai/paper/2404.09486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09486"}},"official":{"repos":["happylkx/mmcode","likaixin2000/mmcode"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/codeeditorbench-evaluating-code-editing","slug":"codeeditorbench-evaluating-code-editing","title":"CodeEditorBench: Evaluating Code Editing Capability of Large Language Models","date":"2024-04-04","arxiv_id":"2404.03543","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codeeditorbench-evaluating-code-editing#ran","syntology_url":"https://syntology.ai/paper/2404.03543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03543"}},"official":null}},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-organized-agents-a-llm-multi-agent","slug":"self-organized-agents-a-llm-multi-agent","title":"Self-Organized Agents: A LLM Multi-Agent Framework toward Ultra Large-Scale Code Generation and Optimization","date":"2024-04-02","arxiv_id":"2404.02183","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-organized-agents-a-llm-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2404.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02183"}},"official":{"repos":["tsukushiai/self-organized-agent"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evocodebench-an-evolving-code-generation","slug":"evocodebench-an-evolving-code-generation","title":"EvoCodeBench: An Evolving Code Generation Benchmark Aligned with Real-World Code Repositories","date":"2024-03-31","arxiv_id":"2404.00599","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 3 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evocodebench-an-evolving-code-generation#ran","syntology_url":"https://syntology.ai/paper/2404.00599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00599"}},"official":{"repos":["seketeam/evocodebench"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/top-leaderboard-ranking-top-coding","slug":"top-leaderboard-ranking-top-coding","title":"Top Leaderboard Ranking = Top Coding Proficiency, Always? EvoEval: Evolving Coding Benchmarks via LLM","date":"2024-03-28","arxiv_id":"2403.19114","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/top-leaderboard-ranking-top-coding#ran","syntology_url":"https://syntology.ai/paper/2403.19114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19114"}},"official":{"repos":["evo-eval/evoeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/devbench-a-comprehensive-benchmark-for","slug":"devbench-a-comprehensive-benchmark-for","title":"Prompting Large Language Models to Tackle the Full Software Development Lifecycle: A Case Study","date":"2024-03-13","arxiv_id":"2403.08604","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/devbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2403.08604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08604"}},"official":{"repos":["open-compass/devbench","open-compass/deveval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inficoder-eval-systematically-evaluating-the","slug":"inficoder-eval-systematically-evaluating-the","title":"InfiBench: Evaluating the Question-Answering Capabilities of Code Large Language Models","date":"2024-03-11","arxiv_id":"2404.07940","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/inficoder-eval-systematically-evaluating-the#ran","syntology_url":"https://syntology.ai/paper/2404.07940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07940"}},"official":{"repos":["infi-coder/infibench-evaluation-harness","infi-coder/infibench-evaluator"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/promoai-process-modeling-with-generative-ai","slug":"promoai-process-modeling-with-generative-ai","title":"ProMoAI: Process Modeling with Generative AI","date":"2024-03-07","arxiv_id":"2403.04327","repositories_listed":0,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/promoai-process-modeling-with-generative-ai#ran","syntology_url":"https://syntology.ai/paper/2403.04327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04327"}},"official":null}},{"url":"/paper/quantifying-contamination-in-evaluating-code","slug":"quantifying-contamination-in-evaluating-code","title":"Quantifying Contamination in Evaluating Code Generation Capabilities of Language Models","date":"2024-03-06","arxiv_id":"2403.04811","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-contamination-in-evaluating-code#ran","syntology_url":"https://syntology.ai/paper/2403.04811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04811"}},"official":{"repos":["yale-nlp/code-llm-contamination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/daco-towards-application-driven-and","slug":"daco-towards-application-driven-and","title":"DACO: Towards Application-Driven and Comprehensive Data Analysis via Code Generation","date":"2024-03-04","arxiv_id":"2403.02528","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/daco-towards-application-driven-and#ran","syntology_url":"https://syntology.ai/paper/2403.02528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02528"}},"official":{"repos":["shirley-wu/daco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-assistant-toolkit-version-2","slug":"open-assistant-toolkit-version-2","title":"Open Assistant Toolkit -- version 2","date":"2024-03-01","arxiv_id":"2403.00586","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-assistant-toolkit-version-2#ran","syntology_url":"https://syntology.ai/paper/2403.00586","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00586"}},"official":{"repos":["grill-lab/oat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-information-refinement-training","slug":"unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","arxiv_id":"2402.18150","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-information-refinement-training#ran","syntology_url":"https://syntology.ai/paper/2402.18150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18150"}},"official":{"repos":["xsc1234/info-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-data-science-agents","slug":"benchmarking-data-science-agents","title":"Benchmarking Data Science Agents","date":"2024-02-27","arxiv_id":"2402.17168","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-science-agents#ran","syntology_url":"https://syntology.ai/paper/2402.17168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17168"}},"official":{"repos":["metacopilot/dseval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ds-agent-automated-data-science-by-empowering","slug":"ds-agent-automated-data-science-by-empowering","title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","date":"2024-02-27","arxiv_id":"2402.17453","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ds-agent-automated-data-science-by-empowering#ran","syntology_url":"https://syntology.ai/paper/2402.17453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17453"}},"official":{"repos":["guosyjlu/ds-agent"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/repoagent-an-llm-powered-open-source","slug":"repoagent-an-llm-powered-open-source","title":"RepoAgent: An LLM-Powered Open-Source Framework for Repository-level Code Documentation Generation","date":"2024-02-26","arxiv_id":"2402.16667","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/repoagent-an-llm-powered-open-source#ran","syntology_url":"https://syntology.ai/paper/2402.16667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16667"}},"official":{"repos":["openbmb/repoagent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ldb-a-large-language-model-debugger-via","slug":"ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","arxiv_id":"2402.16906","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ldb-a-large-language-model-debugger-via#ran","syntology_url":"https://syntology.ai/paper/2402.16906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16906"}},"official":{"repos":["floridsleeves/llmdebugger"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opencodeinterpreter-integrating-code#ran","syntology_url":"https://syntology.ai/paper/2402.14658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14658"}},"official":null}},{"url":"/paper/q-probe-a-lightweight-approach-to-reward","slug":"q-probe-a-lightweight-approach-to-reward","title":"Q-Probe: A Lightweight Approach to Reward Maximization for Language Models","date":"2024-02-22","arxiv_id":"2402.14688","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-probe-a-lightweight-approach-to-reward#ran","syntology_url":"https://syntology.ai/paper/2402.14688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14688"}},"official":{"repos":["likenneth/q_probe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/humaneval-on-latest-gpt-models-2024#ran","syntology_url":"https://syntology.ai/paper/2402.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14852"}},"official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/matplotagent-method-and-evaluation-for-llm","slug":"matplotagent-method-and-evaluation-for-llm","title":"MatPlotAgent: Method and Evaluation for LLM-Based Agentic Scientific Data Visualization","date":"2024-02-18","arxiv_id":"2402.11453","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matplotagent-method-and-evaluation-for-llm#ran","syntology_url":"https://syntology.ai/paper/2402.11453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11453"}},"official":{"repos":["thunlp/matplotagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codemind-a-framework-to-challenge-large","slug":"codemind-a-framework-to-challenge-large","title":"CodeMind: Evaluating Large Language Models for Code Reasoning","date":"2024-02-15","arxiv_id":"2402.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/codemind-a-framework-to-challenge-large#ran","syntology_url":"https://syntology.ai/paper/2402.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09664"}},"official":{"repos":["intelligent-cat-lab/codemind"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-tuning-for-secure-code-generation","slug":"instruction-tuning-for-secure-code-generation","title":"Instruction Tuning for Secure Code Generation","date":"2024-02-14","arxiv_id":"2402.09497","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instruction-tuning-for-secure-code-generation#ran","syntology_url":"https://syntology.ai/paper/2402.09497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09497"}},"official":{"repos":["eth-sri/safecoder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mercury-an-efficiency-benchmark-for-llm-code","slug":"mercury-an-efficiency-benchmark-for-llm-code","title":"Mercury: A Code Efficiency Benchmark for Code Large Language Models","date":"2024-02-12","arxiv_id":"2402.07844","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mercury-an-efficiency-benchmark-for-llm-code#ran","syntology_url":"https://syntology.ai/paper/2402.07844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07844"}},"official":{"repos":["elfsong/mercury"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/entropy-regularized-token-level-policy#ran","syntology_url":"https://syntology.ai/paper/2402.06700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06700"}},"official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/limits-of-transformer-language-models-on","slug":"limits-of-transformer-language-models-on","title":"Limits of Transformer Language Models on Learning to Compose Algorithms","date":"2024-02-08","arxiv_id":"2402.05785","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/limits-of-transformer-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2402.05785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05785"}},"official":{"repos":["ibm/limitations-lm-algorithmic-compositional-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effibench-benchmarking-the-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2402.02037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02037"}},"official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/getting-the-most-out-of-your-tokenizer-for","slug":"getting-the-most-out-of-your-tokenizer-for","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","date":"2024-02-01","arxiv_id":"2402.01035","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/getting-the-most-out-of-your-tokenizer-for#ran","syntology_url":"https://syntology.ai/paper/2402.01035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01035"}},"official":{"repos":["gautierdag/tokenizer-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ppm-automated-generation-of-diverse","slug":"ppm-automated-generation-of-diverse","title":"PPM: Automated Generation of Diverse Programming Problems for Benchmarking Code Generation Models","date":"2024-01-28","arxiv_id":"2401.15545","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ppm-automated-generation-of-diverse#ran","syntology_url":"https://syntology.ai/paper/2401.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15545"}},"official":{"repos":["seekingdream/ppm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-coder-when-the-large-language-model","slug":"deepseek-coder-when-the-large-language-model","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","date":"2024-01-25","arxiv_id":"2401.14196","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepseek-coder-when-the-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2401.14196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14196"}},"official":{"repos":["deepseek-ai/DeepSeek-Coder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-write-parallel-code","slug":"can-large-language-models-write-parallel-code","title":"Can Large Language Models Write Parallel Code?","date":"2024-01-23","arxiv_id":"2401.12554","repositories_listed":1,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-large-language-models-write-parallel-code#ran","syntology_url":"https://syntology.ai/paper/2401.12554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12554"}},"official":{"repos":["parallelcodefoundry/ParEval"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-fusion-of-large-language-models","slug":"knowledge-fusion-of-large-language-models","title":"Knowledge Fusion of Large Language Models","date":"2024-01-19","arxiv_id":"2401.10491","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-fusion-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.10491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10491"}},"official":{"repos":["fanqiwan/fusellm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/langprop-a-code-optimization-framework-using","slug":"langprop-a-code-optimization-framework-using","title":"LangProp: A code optimization framework using Large Language Models applied to driving","date":"2024-01-18","arxiv_id":"2401.10314","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/langprop-a-code-optimization-framework-using#ran","syntology_url":"https://syntology.ai/paper/2401.10314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10314"}},"official":{"repos":["shuishida/langprop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/code-generation-with-alphacodium-from-prompt","slug":"code-generation-with-alphacodium-from-prompt","title":"Code Generation with AlphaCodium: From Prompt Engineering to Flow Engineering","date":"2024-01-16","arxiv_id":"2401.08500","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-generation-with-alphacodium-from-prompt#ran","syntology_url":"https://syntology.ai/paper/2401.08500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08500"}},"official":{"repos":["codium-ai/alphacodium"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ehragent-code-empowers-large-language-models","slug":"ehragent-code-empowers-large-language-models","title":"EHRAgent: Code Empowers Large Language Models for Few-shot Complex Tabular Reasoning on Electronic Health Records","date":"2024-01-13","arxiv_id":"2401.07128","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ehragent-code-empowers-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07128"}},"official":{"repos":["wshi83/ehragent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/oop-object-oriented-programming-evaluation","slug":"oop-object-oriented-programming-evaluation","title":"OOP: Object-Oriented Programming Evaluation Benchmark for Large Language Models","date":"2024-01-12","arxiv_id":"2401.06628","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/oop-object-oriented-programming-evaluation#ran","syntology_url":"https://syntology.ai/paper/2401.06628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06628"}},"official":{"repos":["alphadl/oop-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-the-code-a-simple-method-for-large","slug":"rewriting-the-code-a-simple-method-for-large","title":"Rewriting the Code: A Simple Method for Large Language Model Augmented Code Search","date":"2024-01-09","arxiv_id":"2401.04514","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rewriting-the-code-a-simple-method-for-large#ran","syntology_url":"https://syntology.ai/paper/2401.04514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04514"}},"official":{"repos":["alex-haochenli/reco"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/debugbench-evaluating-debugging-capability-of","slug":"debugbench-evaluating-debugging-capability-of","title":"DebugBench: Evaluating Debugging Capability of Large Language Models","date":"2024-01-09","arxiv_id":"2401.04621","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/debugbench-evaluating-debugging-capability-of#ran","syntology_url":"https://syntology.ai/paper/2401.04621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04621"}},"official":{"repos":["thunlp/debugbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mixtral-of-experts#ran","syntology_url":"https://syntology.ai/paper/2401.04088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04088"}},"official":null}},{"url":"/paper/ast-t5-structure-aware-pretraining-for-code","slug":"ast-t5-structure-aware-pretraining-for-code","title":"AST-T5: Structure-Aware Pretraining for Code Generation and Understanding","date":"2024-01-05","arxiv_id":"2401.03003","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ast-t5-structure-aware-pretraining-for-code#ran","syntology_url":"https://syntology.ai/paper/2401.03003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03003"}},"official":{"repos":["gonglinyuan/ast_t5"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-augmented-llms-expanding-capabilities","slug":"llm-augmented-llms-expanding-capabilities","title":"LLM Augmented LLMs: Expanding Capabilities through Composition","date":"2024-01-04","arxiv_id":"2401.02412","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-augmented-llms-expanding-capabilities#ran","syntology_url":"https://syntology.ai/paper/2401.02412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02412"}},"official":null}},{"url":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-b-b-a-triggering-logical-reasoning-failures#ran","syntology_url":"https://syntology.ai/paper/2401.00757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00757"}},"official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/astraios-parameter-efficient-instruction","slug":"astraios-parameter-efficient-instruction","title":"Astraios: Parameter-Efficient Instruction Tuning Code Large Language Models","date":"2024-01-01","arxiv_id":"2401.00788","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/astraios-parameter-efficient-instruction#ran","syntology_url":"https://syntology.ai/paper/2401.00788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00788"}},"official":{"repos":["bigcode-project/astraios","agi-edgerunners/llm-adapters"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/motcoder-elevating-large-language-models-with","slug":"motcoder-elevating-large-language-models-with","title":"MoTCoder: Elevating Large Language Models with Modular of Thought for Challenging Programming Tasks","date":"2023-12-26","arxiv_id":"2312.15960","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/motcoder-elevating-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2312.15960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15960"}},"official":{"repos":["dvlab-research/motcoder"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-powered-hierarchical-language-agent-for","slug":"llm-powered-hierarchical-language-agent-for","title":"LLM-Powered Hierarchical Language Agent for Real-time Human-AI Coordination","date":"2023-12-23","arxiv_id":"2312.15224","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-powered-hierarchical-language-agent-for#ran","syntology_url":"https://syntology.ai/paper/2312.15224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15224"}},"official":{"repos":["HosnLS/Hierarchical-Language-Agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/taco-topics-in-algorithmic-code-generation","slug":"taco-topics-in-algorithmic-code-generation","title":"TACO: Topics in Algorithmic COde generation dataset","date":"2023-12-22","arxiv_id":"2312.14852","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/taco-topics-in-algorithmic-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14852"}},"official":{"repos":["flagopen/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/turbulence-systematically-and-automatically","slug":"turbulence-systematically-and-automatically","title":"Turbulence: Systematically and Automatically Testing Instruction-Tuned Large Language Models for Code","date":"2023-12-22","arxiv_id":"2312.14856","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turbulence-systematically-and-automatically#ran","syntology_url":"https://syntology.ai/paper/2312.14856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14856"}},"official":{"repos":["shahinhonarvar/turbulence-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentcoder-multi-agent-based-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.13010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13010"}},"official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wavecoder-widespread-and-versatile-enhanced","slug":"wavecoder-widespread-and-versatile-enhanced","title":"WaveCoder: Widespread And Versatile Enhancement For Code Large Language Models By Instruction Tuning","date":"2023-12-20","arxiv_id":"2312.14187","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wavecoder-widespread-and-versatile-enhanced#ran","syntology_url":"https://syntology.ai/paper/2312.14187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14187"}},"official":{"repos":["microsoft/wavecoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/magicoder-source-code-is-all-you-need","slug":"magicoder-source-code-is-all-you-need","title":"Magicoder: Empowering Code Generation with OSS-Instruct","date":"2023-12-04","arxiv_id":"2312.02120","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/magicoder-source-code-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2312.02120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02120"}},"official":{"repos":["ise-uiuc/magicoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-infilling-code-generation","slug":"self-infilling-code-generation","title":"Self-Infilling Code Generation","date":"2023-11-29","arxiv_id":"2311.17972","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-infilling-code-generation#ran","syntology_url":"https://syntology.ai/paper/2311.17972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17972"}},"official":{"repos":["LZhengisme/self-infilling"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-risk-control-a-rigorous-framework-for","slug":"prompt-risk-control-a-rigorous-framework-for","title":"Prompt Risk Control: A Rigorous Framework for Responsible Deployment of Large Language Models","date":"2023-11-22","arxiv_id":"2311.13628","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-risk-control-a-rigorous-framework-for#ran","syntology_url":"https://syntology.ai/paper/2311.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13628"}},"official":{"repos":["thomaspzollo/prompt_risk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ml-bench-large-language-models-leverage-open","slug":"ml-bench-large-language-models-leverage-open","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","date":"2023-11-16","arxiv_id":"2311.09835","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ml-bench-large-language-models-leverage-open#ran","syntology_url":"https://syntology.ai/paper/2311.09835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09835"}},"official":{"repos":["gersteinlab/ml-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-modular-approaches-for-visual","slug":"analyzing-modular-approaches-for-visual","title":"Analyzing Modular Approaches for Visual Question Decomposition","date":"2023-11-10","arxiv_id":"2311.06411","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-modular-approaches-for-visual#ran","syntology_url":"https://syntology.ai/paper/2311.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06411"}},"official":{"repos":["brown-palm/visual-question-decomposition"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/data-augmentation-for-code-translation-with","slug":"data-augmentation-for-code-translation-with","title":"Data Augmentation for Code Translation with Comparable Corpora and Multiple References","date":"2023-11-01","arxiv_id":"2311.00317","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/data-augmentation-for-code-translation-with#ran","syntology_url":"https://syntology.ai/paper/2311.00317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00317"}},"official":{"repos":["veronicium/cmtrans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/lilo-learning-interpretable-libraries-by","slug":"lilo-learning-interpretable-libraries-by","title":"LILO: Learning Interpretable Libraries by Compressing and Documenting Code","date":"2023-10-30","arxiv_id":"2310.19791","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lilo-learning-interpretable-libraries-by#ran","syntology_url":"https://syntology.ai/paper/2310.19791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19791"}},"official":{"repos":["gabegrand/lilo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/symbolic-planning-and-code-generation-for","slug":"symbolic-planning-and-code-generation-for","title":"Symbolic Planning and Code Generation for Grounded Dialogue","date":"2023-10-26","arxiv_id":"2310.17140","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/symbolic-planning-and-code-generation-for#ran","syntology_url":"https://syntology.ai/paper/2310.17140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17140"}},"official":{"repos":["justinchiu/onecommon-gpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/white-box-compiler-fuzzing-empowered-by-large","slug":"white-box-compiler-fuzzing-empowered-by-large","title":"WhiteFox: White-Box Compiler Fuzzing Empowered by Large Language Models","date":"2023-10-24","arxiv_id":"2310.15991","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/white-box-compiler-fuzzing-empowered-by-large#ran","syntology_url":"https://syntology.ai/paper/2310.15991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15991"}},"official":{"repos":["ise-uiuc/whitefox"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-accuracy-evaluating-self-consistency","slug":"beyond-accuracy-evaluating-self-consistency","title":"Beyond Accuracy: Evaluating Self-Consistency of Code Large Language Models with IdentityChain","date":"2023-10-21","arxiv_id":"2310.14053","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":2,"n_ran_checked":4,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/beyond-accuracy-evaluating-self-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.14053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14053"}},"official":{"repos":["marcusm117/IdentityChain"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/codechain-towards-modular-code-generation","slug":"codechain-towards-modular-code-generation","title":"CodeChain: Towards Modular Code Generation Through Chain of Self-revisions with Representative Sub-modules","date":"2023-10-13","arxiv_id":"2310.08992","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codechain-towards-modular-code-generation#ran","syntology_url":"https://syntology.ai/paper/2310.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08992"}},"official":{"repos":["SalesforceAIResearch/CodeChain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-for-scientific","slug":"large-language-models-for-scientific","title":"Large Language Models for Scientific Synthesis, Inference and Explanation","date":"2023-10-12","arxiv_id":"2310.07984","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-for-scientific#ran","syntology_url":"https://syntology.ai/paper/2310.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07984"}},"official":{"repos":["zyzisastudyreallyhardguy/llm4sd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/swe-bench-can-language-models-resolve-real","slug":"swe-bench-can-language-models-resolve-real","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","date":"2023-10-10","arxiv_id":"2310.06770","repositories_listed":8,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/swe-bench-can-language-models-resolve-real#ran","syntology_url":"https://syntology.ai/paper/2310.06770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06770"}},"official":null}},{"url":"/paper/mistral-7b","slug":"mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mistral-7b#ran","syntology_url":"https://syntology.ai/paper/2310.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06825"}},"official":{"repos":["mistralai/mistral-src"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/how-abilities-in-large-language-models-are","slug":"how-abilities-in-large-language-models-are","title":"How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition","date":"2023-10-09","arxiv_id":"2310.05492","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-abilities-in-large-language-models-are#ran","syntology_url":"https://syntology.ai/paper/2310.05492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05492"}},"official":{"repos":["ofa-sys/gsm8k-screl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-agent-tree-search-unifies-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.04406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04406"}},"official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instance-needs-more-care-rewriting-prompts#ran","syntology_url":"https://syntology.ai/paper/2310.02107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02107"}},"official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-taught-optimizer-stop-recursively-self","slug":"self-taught-optimizer-stop-recursively-self","title":"Self-Taught Optimizer (STOP): Recursively Self-Improving Code Generation","date":"2023-10-03","arxiv_id":"2310.02304","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-taught-optimizer-stop-recursively-self#ran","syntology_url":"https://syntology.ai/paper/2310.02304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02304"}},"official":{"repos":["microsoft/stop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gensim-generating-robotic-simulation-tasks","slug":"gensim-generating-robotic-simulation-tasks","title":"GenSim: Generating Robotic Simulation Tasks via Large Language Models","date":"2023-10-02","arxiv_id":"2310.01361","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gensim-generating-robotic-simulation-tasks#ran","syntology_url":"https://syntology.ai/paper/2310.01361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01361"}},"official":{"repos":["liruiw/gensim"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/l2mac-large-language-model-automatic-computer","slug":"l2mac-large-language-model-automatic-computer","title":"L2MAC: Large Language Model Automatic Computer for Extensive Code Generation","date":"2023-10-02","arxiv_id":"2310.02003","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/l2mac-large-language-model-automatic-computer#ran","syntology_url":"https://syntology.ai/paper/2310.02003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02003"}},"official":{"repos":["samholt/l2mac","vanderschaarlab/l2mac"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/enhancing-large-language-models-in-coding#ran","syntology_url":"https://syntology.ai/paper/2309.17272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17272"}},"official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/are-human-generated-demonstrations-necessary","slug":"are-human-generated-demonstrations-necessary","title":"Are Human-generated Demonstrations Necessary for In-context Learning?","date":"2023-09-26","arxiv_id":"2309.14681","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/are-human-generated-demonstrations-necessary#ran","syntology_url":"https://syntology.ai/paper/2309.14681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14681"}},"official":{"repos":["ruili33/sec"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/code-style-in-context-learning-for-knowledge","slug":"code-style-in-context-learning-for-knowledge","title":"Code-Style In-Context Learning for Knowledge-Based Question Answering","date":"2023-09-09","arxiv_id":"2309.04695","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-style-in-context-learning-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2309.04695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04695"}},"official":{"repos":["arthurizijar/kb-coder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-code-generation-by-dynamic","slug":"improving-code-generation-by-dynamic","title":"Hot or Cold? Adaptive Temperature Sampling for Code Generation with Large Language Models","date":"2023-09-06","arxiv_id":"2309.02772","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-code-generation-by-dynamic#ran","syntology_url":"https://syntology.ai/paper/2309.02772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02772"}},"official":{"repos":["lj2lijia/adapt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/copiloting-the-copilots-fusing-large-language","slug":"copiloting-the-copilots-fusing-large-language","title":"Copiloting the Copilots: Fusing Large Language Models with Completion Engines for Automated Program Repair","date":"2023-09-01","arxiv_id":"2309.00608","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/copiloting-the-copilots-fusing-large-language#ran","syntology_url":"https://syntology.ai/paper/2309.00608","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00608"}},"official":{"repos":["ise-uiuc/Repilot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biocoder-a-benchmark-for-bioinformatics-code","slug":"biocoder-a-benchmark-for-bioinformatics-code","title":"BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models","date":"2023-08-31","arxiv_id":"2308.16458","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/biocoder-a-benchmark-for-bioinformatics-code#ran","syntology_url":"https://syntology.ai/paper/2308.16458","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16458"}},"official":{"repos":["gersteinlab/biocoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-llama-open-foundation-models-for-code","slug":"code-llama-open-foundation-models-for-code","title":"Code Llama: Open Foundation Models for Code","date":"2023-08-24","arxiv_id":"2308.12950","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/code-llama-open-foundation-models-for-code#ran","syntology_url":"https://syntology.ai/paper/2308.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12950"}},"official":{"repos":["facebookresearch/codellama"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-parameter-efficient-fine-tuning","slug":"exploring-parameter-efficient-fine-tuning","title":"Exploring Parameter-Efficient Fine-Tuning Techniques for Code Generation with Large Language Models","date":"2023-08-21","arxiv_id":"2308.10462","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-parameter-efficient-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2308.10462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10462"}},"official":{"repos":["martin-wey/peft-llm-code"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"9611e6acc77d39fd4072f05a8efed7487a46b7a6cbb669a80bb12f968137916f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}