{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/code-generation/papers/6","list_of":"/task/code-generation","task":"Code Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":17,"rows_per_page":100,"rows":[501,600],"of":1697,"counts":{"archive_papers_tagged":1697,"with_a_code_link":745,"where_syntology_ran_a_sample":280,"not_listed_spam_title":0,"listed":1697,"listed_where_code_ran":280,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":238,"every_run_a_failure_of_syntologys_instrument":42,"listed_with_a_run_with_no_instrument_failure":238,"listed_every_run_a_failure_of_syntologys_instrument":42,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/code-generation","prev":"/task/code-generation/papers/5","next":"/task/code-generation/papers/7","papers":[{"url":"/paper/humaneval-xl-a-multilingual-code-generation","slug":"humaneval-xl-a-multilingual-code-generation","title":"HumanEval-XL: A Multilingual Code Generation Benchmark for Cross-lingual Natural Language Generalization","date":"2024-02-26","arxiv_id":"2402.16694","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/humaneval-xl-a-multilingual-code-generation#ran","syntology_url":"https://syntology.ai/paper/2402.16694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16694"}},"official":{"repos":["FloatAI/HumanEval-XL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/repoagent-an-llm-powered-open-source","slug":"repoagent-an-llm-powered-open-source","title":"RepoAgent: An LLM-Powered Open-Source Framework for Repository-level Code Documentation Generation","date":"2024-02-26","arxiv_id":"2402.16667","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/repoagent-an-llm-powered-open-source#ran","syntology_url":"https://syntology.ai/paper/2402.16667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16667"}},"official":{"repos":["openbmb/repoagent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ldb-a-large-language-model-debugger-via","slug":"ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","arxiv_id":"2402.16906","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ldb-a-large-language-model-debugger-via#ran","syntology_url":"https://syntology.ai/paper/2402.16906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16906"}},"official":{"repos":["floridsleeves/llmdebugger"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-do-humans-write-code-large-models-do-it","slug":"how-do-humans-write-code-large-models-do-it","title":"How Do Humans Write Code? Large Models Do It the Same Way Too","date":"2024-02-24","arxiv_id":"2402.15729","repositories_listed":1,"syntology":null},{"url":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/opencodeinterpreter-integrating-code#ran","syntology_url":"https://syntology.ai/paper/2402.14658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14658"}},"official":null}},{"url":"/paper/q-probe-a-lightweight-approach-to-reward","slug":"q-probe-a-lightweight-approach-to-reward","title":"Q-Probe: A Lightweight Approach to Reward Maximization for Language Models","date":"2024-02-22","arxiv_id":"2402.14688","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-probe-a-lightweight-approach-to-reward#ran","syntology_url":"https://syntology.ai/paper/2402.14688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14688"}},"official":{"repos":["likenneth/q_probe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/humaneval-on-latest-gpt-models-2024#ran","syntology_url":"https://syntology.ai/paper/2402.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14852"}},"official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rocode-a-dataset-for-measuring-code","slug":"rocode-a-dataset-for-measuring-code","title":"RoCode: A Dataset for Measuring Code Intelligence from Problem Definitions in Romanian","date":"2024-02-20","arxiv_id":"2402.13222","repositories_listed":1,"syntology":null},{"url":"/paper/arks-active-retrieval-in-knowledge-soup-for","slug":"arks-active-retrieval-in-knowledge-soup-for","title":"EVOR: Evolving Retrieval for Code Generation","date":"2024-02-19","arxiv_id":"2402.12317","repositories_listed":1,"syntology":null},{"url":"/paper/matplotagent-method-and-evaluation-for-llm","slug":"matplotagent-method-and-evaluation-for-llm","title":"MatPlotAgent: Method and Evaluation for LLM-Based Agentic Scientific Data Visualization","date":"2024-02-18","arxiv_id":"2402.11453","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matplotagent-method-and-evaluation-for-llm#ran","syntology_url":"https://syntology.ai/paper/2402.11453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11453"}},"official":{"repos":["thunlp/matplotagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codemind-a-framework-to-challenge-large","slug":"codemind-a-framework-to-challenge-large","title":"CodeMind: Evaluating Large Language Models for Code Reasoning","date":"2024-02-15","arxiv_id":"2402.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/codemind-a-framework-to-challenge-large#ran","syntology_url":"https://syntology.ai/paper/2402.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09664"}},"official":{"repos":["intelligent-cat-lab/codemind"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dolphcoder-echo-locating-code-large-language","slug":"dolphcoder-echo-locating-code-large-language","title":"DolphCoder: Echo-Locating Code Large Language Models with Diverse and Multi-Objective Instruction Tuning","date":"2024-02-14","arxiv_id":"2402.09136","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-tuning-for-secure-code-generation","slug":"instruction-tuning-for-secure-code-generation","title":"Instruction Tuning for Secure Code Generation","date":"2024-02-14","arxiv_id":"2402.09497","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instruction-tuning-for-secure-code-generation#ran","syntology_url":"https://syntology.ai/paper/2402.09497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09497"}},"official":{"repos":["eth-sri/safecoder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mpirigen-mpi-code-generation-through-domain","slug":"mpirigen-mpi-code-generation-through-domain","title":"MPIrigen: MPI Code Generation through Domain-Specific Language Models","date":"2024-02-14","arxiv_id":"2402.09126","repositories_listed":1,"syntology":null},{"url":"/paper/safedecoding-defending-against-jailbreak","slug":"safedecoding-defending-against-jailbreak","title":"SafeDecoding: Defending against Jailbreak Attacks via Safety-Aware Decoding","date":"2024-02-14","arxiv_id":"2402.08983","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/safedecoding-defending-against-jailbreak#ran","syntology_url":"https://syntology.ai/paper/2402.08983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08983"}},"official":{"repos":["uw-nsl/safedecoding"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/mercury-an-efficiency-benchmark-for-llm-code","slug":"mercury-an-efficiency-benchmark-for-llm-code","title":"Mercury: A Code Efficiency Benchmark for Code Large Language Models","date":"2024-02-12","arxiv_id":"2402.07844","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mercury-an-efficiency-benchmark-for-llm-code#ran","syntology_url":"https://syntology.ai/paper/2402.07844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07844"}},"official":{"repos":["elfsong/mercury"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/entropy-regularized-token-level-policy#ran","syntology_url":"https://syntology.ai/paper/2402.06700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06700"}},"official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/limits-of-transformer-language-models-on","slug":"limits-of-transformer-language-models-on","title":"Limits of Transformer Language Models on Learning to Compose Algorithms","date":"2024-02-08","arxiv_id":"2402.05785","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/limits-of-transformer-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2402.05785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05785"}},"official":{"repos":["ibm/limitations-lm-algorithmic-compositional-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effibench-benchmarking-the-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2402.02037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02037"}},"official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stepcoder-improve-code-generation-with","slug":"stepcoder-improve-code-generation-with","title":"StepCoder: Improve Code Generation with Reinforcement Learning from Compiler Feedback","date":"2024-02-02","arxiv_id":"2402.01391","repositories_listed":1,"syntology":null},{"url":"/paper/getting-the-most-out-of-your-tokenizer-for","slug":"getting-the-most-out-of-your-tokenizer-for","title":"Getting the most out of your tokenizer for pre-training and domain adaptation","date":"2024-02-01","arxiv_id":"2402.01035","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/getting-the-most-out-of-your-tokenizer-for#ran","syntology_url":"https://syntology.ai/paper/2402.01035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01035"}},"official":{"repos":["gautierdag/tokenizer-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-usage-and-perceptions","slug":"an-empirical-study-on-usage-and-perceptions","title":"An Empirical Study on Usage and Perceptions of LLMs in a Software Engineering Project","date":"2024-01-29","arxiv_id":"2401.16186","repositories_listed":1,"syntology":null},{"url":"/paper/ppm-automated-generation-of-diverse","slug":"ppm-automated-generation-of-diverse","title":"PPM: Automated Generation of Diverse Programming Problems for Benchmarking Code Generation Models","date":"2024-01-28","arxiv_id":"2401.15545","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ppm-automated-generation-of-diverse#ran","syntology_url":"https://syntology.ai/paper/2401.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15545"}},"official":{"repos":["seekingdream/ppm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eagle-speculative-sampling-requires","slug":"eagle-speculative-sampling-requires","title":"EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty","date":"2024-01-26","arxiv_id":"2401.15077","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-coder-when-the-large-language-model","slug":"deepseek-coder-when-the-large-language-model","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","date":"2024-01-25","arxiv_id":"2401.14196","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepseek-coder-when-the-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2401.14196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14196"}},"official":{"repos":["deepseek-ai/DeepSeek-Coder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/designing-silicon-brains-using-llm-leveraging","slug":"designing-silicon-brains-using-llm-leveraging","title":"Designing Silicon Brains using LLM: Leveraging ChatGPT for Automated Description of a Spiking Neuron Array","date":"2024-01-25","arxiv_id":"2402.10920","repositories_listed":1,"syntology":null},{"url":"/paper/improving-natural-language-capability-of-code","slug":"improving-natural-language-capability-of-code","title":"Improving Natural Language Capability of Code Large Language Model","date":"2024-01-25","arxiv_id":"2401.14242","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-efficacy-of-large-language-1","slug":"investigating-the-efficacy-of-large-language-1","title":"Investigating the Efficacy of Large Language Models for Code Clone Detection","date":"2024-01-24","arxiv_id":"2401.13802","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-write-parallel-code","slug":"can-large-language-models-write-parallel-code","title":"Can Large Language Models Write Parallel Code?","date":"2024-01-23","arxiv_id":"2401.12554","repositories_listed":1,"syntology":{"n":19,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-large-language-models-write-parallel-code#ran","syntology_url":"https://syntology.ai/paper/2401.12554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12554"}},"official":{"repos":["parallelcodefoundry/ParEval"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evolutionary-computation-in-the-era-of-large","slug":"evolutionary-computation-in-the-era-of-large","title":"Evolutionary Computation in the Era of Large Language Model: Survey and Roadmap","date":"2024-01-18","arxiv_id":"2401.10034","repositories_listed":1,"syntology":null},{"url":"/paper/langprop-a-code-optimization-framework-using","slug":"langprop-a-code-optimization-framework-using","title":"LangProp: A code optimization framework using Large Language Models applied to driving","date":"2024-01-18","arxiv_id":"2401.10314","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/langprop-a-code-optimization-framework-using#ran","syntology_url":"https://syntology.ai/paper/2401.10314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10314"}},"official":{"repos":["shuishida/langprop"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/using-llm-such-as-chatgpt-for-designing-and","slug":"using-llm-such-as-chatgpt-for-designing-and","title":"Using LLM such as ChatGPT for Designing and Implementing a RISC Processor: Execution,Challenges and Limitations","date":"2024-01-18","arxiv_id":"2401.10364","repositories_listed":1,"syntology":null},{"url":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stuck-in-the-quicksand-of-numeracy-far-from#ran","syntology_url":"https://syntology.ai/paper/2401.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09395"}},"official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/jumpcoder-go-beyond-autoregressive-coder-via","slug":"jumpcoder-go-beyond-autoregressive-coder-via","title":"JumpCoder: Go Beyond Autoregressive Coder via Online Modification","date":"2024-01-15","arxiv_id":"2401.07870","repositories_listed":1,"syntology":null},{"url":"/paper/ehragent-code-empowers-large-language-models","slug":"ehragent-code-empowers-large-language-models","title":"EHRAgent: Code Empowers Large Language Models for Few-shot Complex Tabular Reasoning on Electronic Health Records","date":"2024-01-13","arxiv_id":"2401.07128","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ehragent-code-empowers-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07128"}},"official":{"repos":["wshi83/ehragent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/between-lines-of-code-unraveling-the-distinct","slug":"between-lines-of-code-unraveling-the-distinct","title":"Between Lines of Code: Unraveling the Distinct Patterns of Machine and Human Programmers","date":"2024-01-12","arxiv_id":"2401.06461","repositories_listed":1,"syntology":null},{"url":"/paper/oop-object-oriented-programming-evaluation","slug":"oop-object-oriented-programming-evaluation","title":"OOP: Object-Oriented Programming Evaluation Benchmark for Large Language Models","date":"2024-01-12","arxiv_id":"2401.06628","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/oop-object-oriented-programming-evaluation#ran","syntology_url":"https://syntology.ai/paper/2401.06628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06628"}},"official":{"repos":["alphadl/oop-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/debugbench-evaluating-debugging-capability-of","slug":"debugbench-evaluating-debugging-capability-of","title":"DebugBench: Evaluating Debugging Capability of Large Language Models","date":"2024-01-09","arxiv_id":"2401.04621","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/debugbench-evaluating-debugging-capability-of#ran","syntology_url":"https://syntology.ai/paper/2401.04621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04621"}},"official":{"repos":["thunlp/debugbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/rewriting-the-code-a-simple-method-for-large","slug":"rewriting-the-code-a-simple-method-for-large","title":"Rewriting the Code: A Simple Method for Large Language Model Augmented Code Search","date":"2024-01-09","arxiv_id":"2401.04514","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rewriting-the-code-a-simple-method-for-large#ran","syntology_url":"https://syntology.ai/paper/2401.04514","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04514"}},"official":{"repos":["alex-haochenli/reco"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llm4plc-harnessing-large-language-models-for","slug":"llm4plc-harnessing-large-language-models-for","title":"LLM4PLC: Harnessing Large Language Models for Verifiable Programming of PLCs in Industrial Control Systems","date":"2024-01-08","arxiv_id":"2401.05443","repositories_listed":1,"syntology":null},{"url":"/paper/ast-t5-structure-aware-pretraining-for-code","slug":"ast-t5-structure-aware-pretraining-for-code","title":"AST-T5: Structure-Aware Pretraining for Code Generation and Understanding","date":"2024-01-05","arxiv_id":"2401.03003","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ast-t5-structure-aware-pretraining-for-code#ran","syntology_url":"https://syntology.ai/paper/2401.03003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03003"}},"official":{"repos":["gonglinyuan/ast_t5"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-augmented-llms-expanding-capabilities","slug":"llm-augmented-llms-expanding-capabilities","title":"LLM Augmented LLMs: Expanding Capabilities through Composition","date":"2024-01-04","arxiv_id":"2401.02412","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-augmented-llms-expanding-capabilities#ran","syntology_url":"https://syntology.ai/paper/2401.02412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02412"}},"official":null}},{"url":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-b-b-a-triggering-logical-reasoning-failures#ran","syntology_url":"https://syntology.ai/paper/2401.00757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00757"}},"official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/motcoder-elevating-large-language-models-with","slug":"motcoder-elevating-large-language-models-with","title":"MoTCoder: Elevating Large Language Models with Modular of Thought for Challenging Programming Tasks","date":"2023-12-26","arxiv_id":"2312.15960","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/motcoder-elevating-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2312.15960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15960"}},"official":{"repos":["dvlab-research/motcoder"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-fusion-advancing-prompt-evolution","slug":"instruction-fusion-advancing-prompt-evolution","title":"Instruction Fusion: Advancing Prompt Evolution through Hybridization","date":"2023-12-25","arxiv_id":"2312.15692","repositories_listed":1,"syntology":null},{"url":"/paper/llm-powered-hierarchical-language-agent-for","slug":"llm-powered-hierarchical-language-agent-for","title":"LLM-Powered Hierarchical Language Agent for Real-time Human-AI Coordination","date":"2023-12-23","arxiv_id":"2312.15224","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-powered-hierarchical-language-agent-for#ran","syntology_url":"https://syntology.ai/paper/2312.15224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15224"}},"official":{"repos":["HosnLS/Hierarchical-Language-Agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/automating-the-design-of-multigrid-methods","slug":"automating-the-design-of-multigrid-methods","title":"Automating the Design of Multigrid Methods with Evolutionary Program Synthesis","date":"2023-12-22","arxiv_id":"2312.14875","repositories_listed":1,"syntology":null},{"url":"/paper/taco-topics-in-algorithmic-code-generation","slug":"taco-topics-in-algorithmic-code-generation","title":"TACO: Topics in Algorithmic COde generation dataset","date":"2023-12-22","arxiv_id":"2312.14852","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/taco-topics-in-algorithmic-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.14852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14852"}},"official":{"repos":["flagopen/taco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/turbulence-systematically-and-automatically","slug":"turbulence-systematically-and-automatically","title":"Turbulence: Systematically and Automatically Testing Instruction-Tuned Large Language Models for Code","date":"2023-12-22","arxiv_id":"2312.14856","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turbulence-systematically-and-automatically#ran","syntology_url":"https://syntology.ai/paper/2312.14856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14856"}},"official":{"repos":["shahinhonarvar/turbulence-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentcoder-multi-agent-based-code-generation","slug":"agentcoder-multi-agent-based-code-generation","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","date":"2023-12-20","arxiv_id":"2312.13010","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentcoder-multi-agent-based-code-generation#ran","syntology_url":"https://syntology.ai/paper/2312.13010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13010"}},"official":{"repos":["huangd1999/AgentCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wavecoder-widespread-and-versatile-enhanced","slug":"wavecoder-widespread-and-versatile-enhanced","title":"WaveCoder: Widespread And Versatile Enhancement For Code Large Language Models By Instruction Tuning","date":"2023-12-20","arxiv_id":"2312.14187","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wavecoder-widespread-and-versatile-enhanced#ran","syntology_url":"https://syntology.ai/paper/2312.14187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14187"}},"official":{"repos":["microsoft/wavecoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-a-family-of-highly-capable-multimodal-1","slug":"gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","arxiv_id":"2312.11805","repositories_listed":1,"syntology":null},{"url":"/paper/starvector-generating-scalable-vector","slug":"starvector-generating-scalable-vector","title":"StarVector: Generating Scalable Vector Graphics Code from Images and Text","date":"2023-12-17","arxiv_id":"2312.11556","repositories_listed":1,"syntology":null},{"url":"/paper/lampilot-an-open-benchmark-dataset-for","slug":"lampilot-an-open-benchmark-dataset-for","title":"LaMPilot: An Open Benchmark Dataset for Autonomous Driving with Language Model Programs","date":"2023-12-07","arxiv_id":"2312.04372","repositories_listed":1,"syntology":null},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-infilling-code-generation","slug":"self-infilling-code-generation","title":"Self-Infilling Code Generation","date":"2023-11-29","arxiv_id":"2311.17972","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-infilling-code-generation#ran","syntology_url":"https://syntology.ai/paper/2311.17972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17972"}},"official":{"repos":["LZhengisme/self-infilling"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-for-science-usage-for-code-generation","slug":"llms-for-science-usage-for-code-generation","title":"LLMs for Science: Usage for Code Generation and Data Analysis","date":"2023-11-28","arxiv_id":"2311.16733","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-risk-control-a-rigorous-framework-for","slug":"prompt-risk-control-a-rigorous-framework-for","title":"Prompt Risk Control: A Rigorous Framework for Responsible Deployment of Large Language Models","date":"2023-11-22","arxiv_id":"2311.13628","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prompt-risk-control-a-rigorous-framework-for#ran","syntology_url":"https://syntology.ai/paper/2311.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13628"}},"official":{"repos":["thomaspzollo/prompt_risk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-in-context-learning-of-libraries","slug":"evaluating-in-context-learning-of-libraries","title":"Evaluating In-Context Learning of Libraries for Code Generation","date":"2023-11-16","arxiv_id":"2311.09635","repositories_listed":1,"syntology":null},{"url":"/paper/gencodesearchnet-a-benchmark-test-suite-for","slug":"gencodesearchnet-a-benchmark-test-suite-for","title":"GenCodeSearchNet: A Benchmark Test Suite for Evaluating Generalization in Programming Language Understanding","date":"2023-11-16","arxiv_id":"2311.09707","repositories_listed":1,"syntology":null},{"url":"/paper/intervenor-prompt-the-coding-ability-of-large","slug":"intervenor-prompt-the-coding-ability-of-large","title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","date":"2023-11-16","arxiv_id":"2311.09868","repositories_listed":1,"syntology":null},{"url":"/paper/ml-bench-large-language-models-leverage-open","slug":"ml-bench-large-language-models-leverage-open","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","date":"2023-11-16","arxiv_id":"2311.09835","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ml-bench-large-language-models-leverage-open#ran","syntology_url":"https://syntology.ai/paper/2311.09835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09835"}},"official":{"repos":["gersteinlab/ml-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mela-multilingual-evaluation-of-linguistic","slug":"mela-multilingual-evaluation-of-linguistic","title":"MELA: Multilingual Evaluation of Linguistic Acceptability","date":"2023-11-15","arxiv_id":"2311.09033","repositories_listed":1,"syntology":null},{"url":"/paper/codescope-an-execution-based-multilingual","slug":"codescope-an-execution-based-multilingual","title":"CodeScope: An Execution-based Multilingual Multitask Multidimensional Benchmark for Evaluating LLMs on Code Understanding and Generation","date":"2023-11-14","arxiv_id":"2311.08588","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/codescope-an-execution-based-multilingual#ran","syntology_url":"https://syntology.ai/paper/2311.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08588"}},"official":{"repos":["weixiangyan/codescope"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/language-models-are-better-bug-detector","slug":"language-models-are-better-bug-detector","title":"Language Models are Better Bug Detector Through Code-Pair Classification","date":"2023-11-14","arxiv_id":"2311.07957","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-patch-security-issues","slug":"can-llms-patch-security-issues","title":"Can LLMs Patch Security Issues?","date":"2023-11-13","arxiv_id":"2312.00024","repositories_listed":1,"syntology":null},{"url":"/paper/conceptual-model-interpreter-for-large","slug":"conceptual-model-interpreter-for-large","title":"Conceptual Model Interpreter for Large Language Models","date":"2023-11-11","arxiv_id":"2311.07605","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-modular-approaches-for-visual","slug":"analyzing-modular-approaches-for-visual","title":"Analyzing Modular Approaches for Visual Question Decomposition","date":"2023-11-10","arxiv_id":"2311.06411","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-modular-approaches-for-visual#ran","syntology_url":"https://syntology.ai/paper/2311.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06411"}},"official":{"repos":["brown-palm/visual-question-decomposition"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/cloudeval-yaml-a-practical-benchmark-for","slug":"cloudeval-yaml-a-practical-benchmark-for","title":"CloudEval-YAML: A Practical Benchmark for Cloud Configuration Generation","date":"2023-11-10","arxiv_id":"2401.06786","repositories_listed":1,"syntology":null},{"url":"/paper/assessing-the-promise-and-pitfalls-of-chatgpt","slug":"assessing-the-promise-and-pitfalls-of-chatgpt","title":"Assessing the Promise and Pitfalls of ChatGPT for Automated Code Generation","date":"2023-11-05","arxiv_id":"2311.02640","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-for-code-translation-with","slug":"data-augmentation-for-code-translation-with","title":"Data Augmentation for Code Translation with Comparable Corpora and Multiple References","date":"2023-11-01","arxiv_id":"2311.00317","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/data-augmentation-for-code-translation-with#ran","syntology_url":"https://syntology.ai/paper/2311.00317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00317"}},"official":{"repos":["veronicium/cmtrans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamics-of-instruction-tuning-each-ability","slug":"dynamics-of-instruction-tuning-each-ability","title":"Dynamics of Instruction Tuning: Each Ability of Large Language Models Has Its Own Growth Pace","date":"2023-10-30","arxiv_id":"2310.19651","repositories_listed":1,"syntology":null},{"url":"/paper/lilo-learning-interpretable-libraries-by","slug":"lilo-learning-interpretable-libraries-by","title":"LILO: Learning Interpretable Libraries by Compressing and Documenting Code","date":"2023-10-30","arxiv_id":"2310.19791","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lilo-learning-interpretable-libraries-by#ran","syntology_url":"https://syntology.ai/paper/2310.19791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19791"}},"official":{"repos":["gabegrand/lilo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/personalised-distillation-empowering-open","slug":"personalised-distillation-empowering-open","title":"Personalised Distillation: Empowering Open-Sourced LLMs with Adaptive Learning for Code Generation","date":"2023-10-28","arxiv_id":"2310.18628","repositories_listed":1,"syntology":null},{"url":"/paper/pitfalls-in-language-models-for-code","slug":"pitfalls-in-language-models-for-code","title":"Pitfalls in Language Models for Code Intelligence: A Taxonomy and Survey","date":"2023-10-27","arxiv_id":"2310.17903","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-planning-and-code-generation-for","slug":"symbolic-planning-and-code-generation-for","title":"Symbolic Planning and Code Generation for Grounded Dialogue","date":"2023-10-26","arxiv_id":"2310.17140","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/symbolic-planning-and-code-generation-for#ran","syntology_url":"https://syntology.ai/paper/2310.17140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17140"}},"official":{"repos":["justinchiu/onecommon-gpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spikingjelly-an-open-source-machine-learning","slug":"spikingjelly-an-open-source-machine-learning","title":"SpikingJelly: An open-source machine learning infrastructure platform for spike-based intelligence","date":"2023-10-25","arxiv_id":"2310.16620","repositories_listed":1,"syntology":null},{"url":"/paper/stelocoder-a-decoder-only-llm-for-multi","slug":"stelocoder-a-decoder-only-llm-for-multi","title":"SteloCoder: a Decoder-Only LLM for Multi-Language to Python Code Translation","date":"2023-10-24","arxiv_id":"2310.15539","repositories_listed":1,"syntology":null},{"url":"/paper/white-box-compiler-fuzzing-empowered-by-large","slug":"white-box-compiler-fuzzing-empowered-by-large","title":"WhiteFox: White-Box Compiler Fuzzing Empowered by Large Language Models","date":"2023-10-24","arxiv_id":"2310.15991","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/white-box-compiler-fuzzing-empowered-by-large#ran","syntology_url":"https://syntology.ai/paper/2310.15991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15991"}},"official":{"repos":["ise-uiuc/whitefox"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-accuracy-evaluating-self-consistency","slug":"beyond-accuracy-evaluating-self-consistency","title":"Beyond Accuracy: Evaluating Self-Consistency of Code Large Language Models with IdentityChain","date":"2023-10-21","arxiv_id":"2310.14053","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":2,"n_ran_checked":4,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"10 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/beyond-accuracy-evaluating-self-consistency#ran","syntology_url":"https://syntology.ai/paper/2310.14053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14053"}},"official":{"repos":["marcusm117/IdentityChain"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/codechain-towards-modular-code-generation","slug":"codechain-towards-modular-code-generation","title":"CodeChain: Towards Modular Code Generation Through Chain of Self-revisions with Representative Sub-modules","date":"2023-10-13","arxiv_id":"2310.08992","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/codechain-towards-modular-code-generation#ran","syntology_url":"https://syntology.ai/paper/2310.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08992"}},"official":{"repos":["SalesforceAIResearch/CodeChain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gamegpt-multi-agent-collaborative-framework","slug":"gamegpt-multi-agent-collaborative-framework","title":"GameGPT: Multi-agent Collaborative Framework for Game Development","date":"2023-10-12","arxiv_id":"2310.08067","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-scientific","slug":"large-language-models-for-scientific","title":"Large Language Models for Scientific Synthesis, Inference and Explanation","date":"2023-10-12","arxiv_id":"2310.07984","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-for-scientific#ran","syntology_url":"https://syntology.ai/paper/2310.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07984"}},"official":{"repos":["zyzisastudyreallyhardguy/llm4sd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mini-dalle3-interactive-text-to-image-by","slug":"mini-dalle3-interactive-text-to-image-by","title":"Mini-DALLE3: Interactive Text to Image by Prompting Large Language Models","date":"2023-10-11","arxiv_id":"2310.07653","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-explaining-large-language","slug":"benchmarking-and-explaining-large-language","title":"Benchmarking and Explaining Large Language Model-based Code Generation: A Causality-Centric Approach","date":"2023-10-10","arxiv_id":"2310.06680","repositories_listed":1,"syntology":null},{"url":"/paper/trace-a-comprehensive-benchmark-for-continual","slug":"trace-a-comprehensive-benchmark-for-continual","title":"TRACE: A Comprehensive Benchmark for Continual Learning in Large Language Models","date":"2023-10-10","arxiv_id":"2310.06762","repositories_listed":1,"syntology":null},{"url":"/paper/what-if-the-tv-was-off-examining","slug":"what-if-the-tv-was-off-examining","title":"What If the TV Was Off? Examining Counterfactual Reasoning Abilities of Multi-modal Language Models","date":"2023-10-10","arxiv_id":"2310.06627","repositories_listed":1,"syntology":null},{"url":"/paper/llm4vv-developing-llm-driven-testsuite-for","slug":"llm4vv-developing-llm-driven-testsuite-for","title":"LLM4VV: Developing LLM-Driven Testsuite for Compiler Validation","date":"2023-10-08","arxiv_id":"2310.04963","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-llm-agent-network-an-llm-agent","slug":"dynamic-llm-agent-network-an-llm-agent","title":"A Dynamic LLM-Powered Agent Network for Task-Oriented Agent Collaboration","date":"2023-10-03","arxiv_id":"2310.02170","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-llm-agent-network-an-llm-agent#ran","syntology_url":"https://syntology.ai/paper/2310.02170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02170"}},"official":{"repos":["salt-nlp/dylan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/instance-needs-more-care-rewriting-prompts#ran","syntology_url":"https://syntology.ai/paper/2310.02107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02107"}},"official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-taught-optimizer-stop-recursively-self","slug":"self-taught-optimizer-stop-recursively-self","title":"Self-Taught Optimizer (STOP): Recursively Self-Improving Code Generation","date":"2023-10-03","arxiv_id":"2310.02304","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-taught-optimizer-stop-recursively-self#ran","syntology_url":"https://syntology.ai/paper/2310.02304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02304"}},"official":{"repos":["microsoft/stop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gensim-generating-robotic-simulation-tasks","slug":"gensim-generating-robotic-simulation-tasks","title":"GenSim: Generating Robotic Simulation Tasks via Large Language Models","date":"2023-10-02","arxiv_id":"2310.01361","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gensim-generating-robotic-simulation-tasks#ran","syntology_url":"https://syntology.ai/paper/2310.01361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01361"}},"official":{"repos":["liruiw/gensim"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/enhancing-large-language-models-in-coding#ran","syntology_url":"https://syntology.ai/paper/2309.17272","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17272"}},"official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/are-human-generated-demonstrations-necessary","slug":"are-human-generated-demonstrations-necessary","title":"Are Human-generated Demonstrations Necessary for In-context Learning?","date":"2023-09-26","arxiv_id":"2309.14681","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/are-human-generated-demonstrations-necessary#ran","syntology_url":"https://syntology.ai/paper/2309.14681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14681"}},"official":{"repos":["ruili33/sec"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openai-s-gpt4-as-coding-assistant","slug":"openai-s-gpt4-as-coding-assistant","title":"OpenAi's GPT4 as coding assistant","date":"2023-09-22","arxiv_id":"2309.12732","repositories_listed":1,"syntology":null},{"url":"/paper/openchat-advancing-open-source-language","slug":"openchat-advancing-open-source-language","title":"OpenChat: Advancing Open-source Language Models with Mixed-Quality Data","date":"2023-09-20","arxiv_id":"2309.11235","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/openchat-advancing-open-source-language#ran","syntology_url":"https://syntology.ai/paper/2309.11235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11235"}},"official":{"repos":["imoneoi/openchat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/layoutnuwa-revealing-the-hidden-layout","slug":"layoutnuwa-revealing-the-hidden-layout","title":"LayoutNUWA: Revealing the Hidden Layout Expertise of Large Language Models","date":"2023-09-18","arxiv_id":"2309.09506","repositories_listed":1,"syntology":null},{"url":"/paper/verilogeval-evaluating-large-language-models","slug":"verilogeval-evaluating-large-language-models","title":"VerilogEval: Evaluating Large Language Models for Verilog Code Generation","date":"2023-09-14","arxiv_id":"2309.07544","repositories_listed":1,"syntology":null},{"url":"/paper/ungar-unicode-x2013-a-c-framework-for-real","slug":"ungar-unicode-x2013-a-c-framework-for-real","title":"Ungar $\\unicode{x2013}$ A C++ Framework for Real-Time Optimal Control Using Template Metaprogramming","date":"2023-09-13","arxiv_id":"2309.06783","repositories_listed":1,"syntology":null}],"record_sha256":"8326a1f69d86efd170a0728825681f3831b71e91b383b0fb58a505b789d6079b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}