{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/ran/3","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":526,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4/papers/ran/1","prev":"/method/gpt-4/papers/ran/2","next":"/method/gpt-4/papers/ran/4","papers":[{"paper":"/paper/exploring-safety-generalization-challenges-of","slug":"exploring-safety-generalization-challenges-of","title":"CodeAttack: Revealing Safety Generalization Challenges of Large Language Models via Code Completion","date":"2024-03-12","arxiv_id":"2403.07865","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["renqibing/CodeAttack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/training-small-multimodal-models-to-bridge","slug":"training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","arxiv_id":"2403.08002","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/llava-rad"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/clinicalmamba-a-generative-clinical-language","slug":"clinicalmamba-a-generative-clinical-language","title":"ClinicalMamba: A Generative Clinical Language Model on Longitudinal Clinical Notes","date":"2024-03-09","arxiv_id":"2403.05795","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whaleloops/clinicalmamba"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoeval-done-right-using-synthetic-data-for","slug":"autoeval-done-right-using-synthetic-data-for","title":"AutoEval Done Right: Using Synthetic Data for Model Evaluation","date":"2024-03-09","arxiv_id":"2403.07008","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pierreboyeau/autoeval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-t-remember-details-in-long-documents-you","slug":"can-t-remember-details-in-long-documents-you","title":"Can't Remember Details in Long Documents? You Need Some R&R","date":"2024-03-08","arxiv_id":"2403.05004","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["casetext/r-and-r"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/erbench-an-entity-relationship-based","slug":"erbench-an-entity-relationship-based","title":"ERBench: An Entity-Relationship based Automatically Verifiable Hallucination Benchmark for Large Language Models","date":"2024-03-08","arxiv_id":"2403.05266","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dilab-kaist/erbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm4decompile-decompiling-binary-code-with","slug":"llm4decompile-decompiling-binary-code-with","title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","date":"2024-03-08","arxiv_id":"2403.05286","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["albertan017/LLM4Decompile"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/llms-in-the-imaginarium-tool-learning-through","slug":"llms-in-the-imaginarium-tool-learning-through","title":"LLMs in the Imaginarium: Tool Learning through Simulated Trial and Error","date":"2024-03-07","arxiv_id":"2403.04746","n_code_links":1,"syntology":{"ran":6,"of":13,"n_ran_checked":6,"n_instrument":0,"unverified":7,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["microsoft/simulated-trial-and-error"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/injecagent-benchmarking-indirect-prompt","slug":"injecagent-benchmarking-indirect-prompt","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated Large Language Model Agents","date":"2024-03-05","arxiv_id":"2403.02691","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uiuc-kang-lab/injecagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm","slug":"an-empirical-study-of-llm-as-a-judge-for-llm","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation: Fine-tuned Judge Model is not a General Substitute for GPT-4","date":"2024-03-05","arxiv_id":"2403.02839","n_code_links":1,"syntology":{"ran":15,"of":18,"n_ran_checked":15,"n_instrument":0,"unverified":3,"pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["huihuichyan/unlimitedjudge"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/varierr-nli-separating-annotation-error-from","slug":"varierr-nli-separating-annotation-error-from","title":"VariErr NLI: Separating Annotation Error from Human Label Variation","date":"2024-03-04","arxiv_id":"2403.01931","n_code_links":0,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lab-large-scale-alignment-for-chatbots","slug":"lab-large-scale-alignment-for-chatbots","title":"LAB: Large-Scale Alignment for ChatBots","date":"2024-03-02","arxiv_id":"2403.01081","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/improving-the-validity-of-automatically","slug":"improving-the-validity-of-automatically","title":"Improving the Validity of Automatically Generated Feedback via Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01304","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["umass-ml4ed/feedback-gen-dpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/hire-a-linguist-learning-endangered-languages","slug":"hire-a-linguist-learning-endangered-languages","title":"Hire a Linguist!: Learning Endangered Languages with In-Context Linguistic Descriptions","date":"2024-02-28","arxiv_id":"2402.18025","n_code_links":2,"syntology":{"ran":8,"of":13,"n_ran_checked":7,"n_instrument":1,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["leililab/lingollm","llilab/llm4endangeredlang"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/making-them-ask-and-answer-jailbreaking-large","slug":"making-them-ask-and-answer-jailbreaking-large","title":"Making Them Ask and Answer: Jailbreaking Large Language Models in Few Queries via Disguise and Reconstruction","date":"2024-02-28","arxiv_id":"2402.18104","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["llm-dra/dra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clustering-and-ranking-diversity-preserved","slug":"clustering-and-ranking-diversity-preserved","title":"Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation","date":"2024-02-28","arxiv_id":"2402.18191","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ironbeliever/car"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/fofo-a-benchmark-to-evaluate-llms-format","slug":"fofo-a-benchmark-to-evaluate-llms-format","title":"FOFO: A Benchmark to Evaluate LLMs' Format-Following Capability","date":"2024-02-28","arxiv_id":"2402.18667","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["salesforceairesearch/fofo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-llm-generate-culturally-relevant","slug":"can-llm-generate-culturally-relevant","title":"Can LLM Generate Culturally Relevant Commonsense QA Data? Case Study in Indonesian and Sundanese","date":"2024-02-27","arxiv_id":"2402.17302","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rifkiaputri/id-csqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ds-agent-automated-data-science-by-empowering","slug":"ds-agent-automated-data-science-by-empowering","title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","date":"2024-02-27","arxiv_id":"2402.17453","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["guosyjlu/ds-agent"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/songcomposer-a-large-language-model-for-lyric","slug":"songcomposer-a-large-language-model-for-lyric","title":"SongComposer: A Large Language Model for Lyric and Melody Generation in Song Composition","date":"2024-02-27","arxiv_id":"2402.17645","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pjlab-songcomposer/songcomposer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/codes-towards-building-open-source-language","slug":"codes-towards-building-open-source-language","title":"CodeS: Towards Building Open-source Language Models for Text-to-SQL","date":"2024-02-26","arxiv_id":"2402.16347","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":12,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ruckbreasoning/codes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/graphwiz-an-instruction-following-language","slug":"graphwiz-an-instruction-following-language","title":"GraphWiz: An Instruction-Following Language Model for Graph Problems","date":"2024-02-25","arxiv_id":"2402.16029","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nuochenpku/Graph-Reasoning-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ehrnoteqa-a-patient-specific-question","slug":"ehrnoteqa-a-patient-specific-question","title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","date":"2024-02-25","arxiv_id":"2402.16040","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ji-youn-kim/ehrnoteqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/drattack-prompt-decomposition-and","slug":"drattack-prompt-decomposition-and","title":"DrAttack: Prompt Decomposition and Reconstruction Makes Powerful LLM Jailbreakers","date":"2024-02-25","arxiv_id":"2402.16914","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["xirui-li/drattack"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-data-centric-approach-to-generate-faithful","slug":"a-data-centric-approach-to-generate-faithful","title":"A Data-Centric Approach To Generate Faithful and High Quality Patient Summaries with Large Language Models","date":"2024-02-23","arxiv_id":"2402.15422","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stefanhgm/patient_summaries_with_llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","n_code_links":1,"syntology":{"ran":8,"of":17,"n_ran_checked":8,"n_instrument":0,"unverified":9,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/middleware-for-llms-tools-are-instrumental","slug":"middleware-for-llms-tools-are-instrumental","title":"Middleware for LLMs: Tools Are Instrumental for Language Agents in Complex Environments","date":"2024-02-22","arxiv_id":"2402.14672","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["OSU-NLP-Group/Middleware","osu-nlp-group/fuxi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tokenization-counts-the-impact-of","slug":"tokenization-counts-the-impact-of","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","date":"2024-02-22","arxiv_id":"2402.14903","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aadityasingh/tokenizationcounts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-llms-effective-negotiators-systematic","slug":"are-llms-effective-negotiators-systematic","title":"Are LLMs Effective Negotiators? Systematic Evaluation of the Multifaceted Capabilities of LLMs in Negotiation Dialogues","date":"2024-02-21","arxiv_id":"2402.13550","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dsincerity/syseval-negollms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unigraph-learning-a-cross-domain-graph","slug":"unigraph-learning-a-cross-domain-graph","title":"UniGraph: Learning a Unified Cross-Domain Foundation Model for Text-Attributed Graphs","date":"2024-02-21","arxiv_id":"2402.13630","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yf-he/UniGraph"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-building-multilingual-language-model","slug":"towards-building-multilingual-language-model","title":"Towards Building Multilingual Language Model for Medicine","date":"2024-02-21","arxiv_id":"2402.13963","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["magic-ai4med/mmedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fanoutqa-multi-hop-multi-document-question","slug":"fanoutqa-multi-hop-multi-document-question","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.14116","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhudotexe/fanoutqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pca-bench-evaluating-multimodal-large","slug":"pca-bench-evaluating-multimodal-large","title":"PCA-Bench: Evaluating Multimodal Large Language Models in Perception-Cognition-Action Chain","date":"2024-02-21","arxiv_id":"2402.15527","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pkunlp-icler/pca-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-finben-an-holistic-financial-benchmark","slug":"the-finben-an-holistic-financial-benchmark","title":"FinBen: A Holistic Financial Benchmark for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12659","n_code_links":2,"syntology":{"ran":8,"of":12,"n_ran_checked":6,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["the-finai/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/me-llama-foundation-large-language-models-for","slug":"me-llama-foundation-large-language-models-for","title":"Me LLaMA: Foundation Large Language Models for Medical Applications","date":"2024-02-20","arxiv_id":"2402.12749","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bids-xu-lab/me-llama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/artprompt-ascii-art-based-jailbreak-attacks","slug":"artprompt-ascii-art-based-jailbreak-attacks","title":"ArtPrompt: ASCII Art-based Jailbreak Attacks against Aligned LLMs","date":"2024-02-19","arxiv_id":"2402.11753","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":10,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["uw-nsl/ArtPrompt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-clip-unsupervised-adversarial-fine","slug":"robust-clip-unsupervised-adversarial-fine","title":"Robust CLIP: Unsupervised Adversarial Fine-Tuning of Vision Embeddings for Robust Large Vision-Language Models","date":"2024-02-19","arxiv_id":"2402.12336","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":1,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chs20/robustvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/factpico-factuality-evaluation-for-plain","slug":"factpico-factuality-evaluation-for-plain","title":"FactPICO: Factuality Evaluation for Plain Language Summarization of Medical Evidence","date":"2024-02-18","arxiv_id":"2402.11456","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lilywchen/factpico"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-task-inference-can-large-language","slug":"multi-task-inference-can-large-language","title":"Multi-Task Inference: Can Large Language Models Follow Multiple Instructions at Once?","date":"2024-02-18","arxiv_id":"2402.11597","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["guijinson/mti-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-from-failure-integrating-negative","slug":"learning-from-failure-integrating-negative","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","date":"2024-02-18","arxiv_id":"2402.11651","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["reason-wang/nat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/zerog-investigating-cross-dataset-zero-shot","slug":"zerog-investigating-cross-dataset-zero-shot","title":"ZeroG: Investigating Cross-dataset Zero-shot Transferability in Graphs","date":"2024-02-17","arxiv_id":"2402.11235","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nineabyss/zerog"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/jailbreaking-proprietary-large-language","slug":"jailbreaking-proprietary-large-language","title":"When \"Competency\" in Reasoning Opens the Door to Vulnerability: Jailbreaking LLMs via Novel Complex Ciphers","date":"2024-02-16","arxiv_id":"2402.10601","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["divijh/jailbreak_cryptography"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-llms-speak-for-diverse-people-tuning-llms","slug":"can-llms-speak-for-diverse-people-tuning-llms","title":"Can LLMs Speak For Diverse People? Tuning LLMs via Debate to Generate Controllable Controversial Statements","date":"2024-02-16","arxiv_id":"2402.10614","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianyi-lab/debatune"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-engineering-for-scaling-language-models","slug":"data-engineering-for-scaling-language-models","title":"Data Engineering for Scaling Language Models to 128K Context","date":"2024-02-15","arxiv_id":"2402.10171","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/long-context-data-engineering"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-the-authoring-of-autotutors-with","slug":"scaling-the-authoring-of-autotutors-with","title":"AutoTutor meets Large Language Models: A Language Model Tutor with Rich Pedagogy and Guardrails","date":"2024-02-14","arxiv_id":"2402.09216","n_code_links":1,"syntology":{"ran":4,"of":10,"n_ran_checked":4,"n_instrument":0,"unverified":6,"pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["eth-lre/mwptutor"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/llasmol-advancing-large-language-models-for","slug":"llasmol-advancing-large-language-models-for","title":"LlaSMol: Advancing Large Language Models for Chemistry with a Large-Scale, Comprehensive, High-Quality Instruction Tuning Dataset","date":"2024-02-14","arxiv_id":"2402.09391","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["osu-nlp-group/llm4chem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/llaga-large-language-and-graph-assistant","slug":"llaga-large-language-and-graph-assistant","title":"LLaGA: Large Language and Graph Assistant","date":"2024-02-13","arxiv_id":"2402.08170","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chenrunjin/llaga","vita-group/llaga"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/bbox-adapter-lightweight-adapting-for-black","slug":"bbox-adapter-lightweight-adapting-for-black","title":"BBox-Adapter: Lightweight Adapting for Black-Box Large Language Models","date":"2024-02-13","arxiv_id":"2402.08219","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haotiansun14/bbox-adapter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompt-optimization-in-multi-step-tasks","slug":"prompt-optimization-in-multi-step-tasks","title":"PRompt Optimization in Multi-Step Tasks (PROMST): Integrating Human Feedback and Heuristic-based Sampling","date":"2024-02-13","arxiv_id":"2402.08702","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yongchao98/promst"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/ecellm-generalizing-large-language-models-for","slug":"ecellm-generalizing-large-language-models-for","title":"eCeLLM: Generalizing Large Language Models for E-commerce from Large-scale, High-quality Instruction Data","date":"2024-02-13","arxiv_id":"2402.08831","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ninglab/ecellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/addressing-cognitive-bias-in-medical-language","slug":"addressing-cognitive-bias-in-medical-language","title":"Addressing cognitive bias in medical language models","date":"2024-02-12","arxiv_id":"2402.08113","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["carlwharris/cog-bias-med-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/urbankgent-a-unified-large-language-model","slug":"urbankgent-a-unified-large-language-model","title":"UrbanKGent: A Unified Large Language Model Agent Framework for Urban Knowledge Graph Construction","date":"2024-02-10","arxiv_id":"2402.06861","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["usail-hkust/urbankgent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/resumeflow-an-llm-facilitated-pipeline-for","slug":"resumeflow-an-llm-facilitated-pipeline-for","title":"ResumeFlow: An LLM-facilitated Pipeline for Personalized Resume Generation and Refinement","date":"2024-02-09","arxiv_id":"2402.06221","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Ztrimus/job-llm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/culturellm-incorporating-cultural-differences","slug":"culturellm-incorporating-cultural-differences","title":"CultureLLM: Incorporating Cultural Differences into Large Language Models","date":"2024-02-09","arxiv_id":"2402.10946","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["scarelette/culturellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/noise-contrastive-alignment-of-language","slug":"noise-contrastive-alignment-of-language","title":"Noise Contrastive Alignment of Language Models with Explicit Rewards","date":"2024-02-08","arxiv_id":"2402.05369","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-ml/noise-contrastive-alignment"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/limits-of-transformer-language-models-on","slug":"limits-of-transformer-language-models-on","title":"Limits of Transformer Language Models on Learning to Compose Algorithms","date":"2024-02-08","arxiv_id":"2402.05785","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":3,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ibm/limitations-lm-algorithmic-compositional-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-well-can-llms-negotiate-negotiationarena","slug":"how-well-can-llms-negotiate-negotiationarena","title":"How Well Can LLMs Negotiate? NegotiationArena Platform and Analysis","date":"2024-02-08","arxiv_id":"2402.05863","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vinid/negotiationarena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-large-language-model-agents-simulate","slug":"can-large-language-model-agents-simulate","title":"Can Large Language Model Agents Simulate Human Trust Behavior?","date":"2024-02-07","arxiv_id":"2402.04559","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["camel-ai/agent-trust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/long-is-more-for-alignment-a-simple-but-tough","slug":"long-is-more-for-alignment-a-simple-but-tough","title":"Long Is More for Alignment: A Simple but Tough-to-Beat Baseline for Instruction Fine-Tuning","date":"2024-02-07","arxiv_id":"2402.04833","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/long-is-more-for-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/opening-the-ai-black-box-program-synthesis","slug":"opening-the-ai-black-box-program-synthesis","title":"Opening the AI black box: program synthesis via mechanistic interpretability","date":"2024-02-07","arxiv_id":"2402.05110","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ejmichaud/neural-verification"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/anytool-self-reflective-hierarchical-agents","slug":"anytool-self-reflective-hierarchical-agents","title":"AnyTool: Self-Reflective, Hierarchical Agents for Large-Scale API Calls","date":"2024-02-06","arxiv_id":"2402.04253","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dyabel/anytool"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/graph-enhanced-large-language-models-in","slug":"graph-enhanced-large-language-models-in","title":"Graph-enhanced Large Language Models in Asynchronous Plan Reasoning","date":"2024-02-05","arxiv_id":"2402.02805","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":0,"n_instrument":6,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fangru-lin/graph-llm-asynchow-plan"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-mamba-capable-of-in-context-learning","slug":"is-mamba-capable-of-in-context-learning","title":"Is Mamba Capable of In-Context Learning?","date":"2024-02-05","arxiv_id":"2402.03170","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["automl/is_mamba_capable_of_icl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/deepseekmath-pushing-the-limits-of","slug":"deepseekmath-pushing-the-limits-of","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","date":"2024-02-05","arxiv_id":"2402.03300","n_code_links":5,"syntology":{"ran":14,"of":24,"n_ran_checked":10,"n_instrument":4,"unverified":10,"pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 10 unverified","official":{"repos":["deepseek-ai/deepseek-math"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/travelplanner-a-benchmark-for-real-world","slug":"travelplanner-a-benchmark-for-real-world","title":"TravelPlanner: A Benchmark for Real-World Planning with Language Agents","date":"2024-02-02","arxiv_id":"2402.01622","n_code_links":2,"syntology":{"ran":17,"of":17,"n_ran_checked":13,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["OSU-NLP-Group/TravelPlanner"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/raptor-recursive-abstractive-processing-for","slug":"raptor-recursive-abstractive-processing-for","title":"RAPTOR: Recursive Abstractive Processing for Tree-Organized Retrieval","date":"2024-01-31","arxiv_id":"2401.18059","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["parthsarthi03/RAPTOR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mt-eval-a-multi-turn-capabilities-evaluation","slug":"mt-eval-a-multi-turn-capabilities-evaluation","title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models","date":"2024-01-30","arxiv_id":"2401.16745","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwanwaichung/mt-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/robust-prompt-optimization-for-defending","slug":"robust-prompt-optimization-for-defending","title":"Robust Prompt Optimization for Defending Language Models Against Jailbreaking Attacks","date":"2024-01-30","arxiv_id":"2401.17263","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lapisrocks/rpo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multihop-rag-benchmarking-retrieval-augmented","slug":"multihop-rag-benchmarking-retrieval-augmented","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","date":"2024-01-27","arxiv_id":"2401.15391","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yixuantt/MultiHop-RAG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chemdfm-dialogue-foundation-model-for","slug":"chemdfm-dialogue-foundation-model-for","title":"ChemDFM: A Large Language Foundation Model for Chemistry","date":"2024-01-26","arxiv_id":"2401.14818","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/prompting-large-language-models-for-zero-shot-1","slug":"prompting-large-language-models-for-zero-shot-1","title":"Prompting Large Language Models for Zero-Shot Clinical Prediction with Structured Longitudinal Electronic Health Record Data","date":"2024-01-25","arxiv_id":"2402.01713","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yhzhu99/llm4healthcare"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/contextual-evaluating-context-sensitive-text","slug":"contextual-evaluating-context-sensitive-text","title":"ConTextual: Evaluating Context-Sensitive Text-Rich Visual Reasoning in Large Multimodal Models","date":"2024-01-24","arxiv_id":"2401.13311","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rohan598/contextual"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-prompting-enhancing-language-models-with","slug":"meta-prompting-enhancing-language-models-with","title":"Meta-Prompting: Enhancing Language Models with Task-Agnostic Scaffolding","date":"2024-01-23","arxiv_id":"2401.12954","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["suzgunmirac/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/args-alignment-as-reward-guided-search","slug":"args-alignment-as-reward-guided-search","title":"ARGS: Alignment as Reward-Guided Search","date":"2024-01-23","arxiv_id":"2402.01694","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["deeplearning-wisc/args"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/inducing-high-energy-latency-of-large-vision","slug":"inducing-high-energy-latency-of-large-vision","title":"Inducing High Energy-Latency of Large Vision-Language Models with Verbose Images","date":"2024-01-20","arxiv_id":"2401.11170","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":0,"n_instrument":5,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kuofenggao/verbose_images"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/badchain-backdoor-chain-of-thought-prompting","slug":"badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","arxiv_id":"2401.12242","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["django-jiang/badchain"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/r-judge-benchmarking-safety-risk-awareness","slug":"r-judge-benchmarking-safety-risk-awareness","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","date":"2024-01-18","arxiv_id":"2401.10019","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lordog/r-judge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-rewarding-language-models","slug":"self-rewarding-language-models","title":"Self-Rewarding Language Models","date":"2024-01-18","arxiv_id":"2401.10020","n_code_links":3,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/augmenting-math-word-problems-via-iterative","slug":"augmenting-math-word-problems-via-iterative","title":"Augmenting Math Word Problems via Iterative Question Composing","date":"2024-01-17","arxiv_id":"2401.09003","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["iiis-ai/iterativequestioncomposing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/stuck-in-the-quicksand-of-numeracy-far-from","slug":"stuck-in-the-quicksand-of-numeracy-far-from","title":"Evaluating LLMs' Mathematical and Coding Competency through Ontology-guided Interventions","date":"2024-01-17","arxiv_id":"2401.09395","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["declare-lab/llm-reasoningtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mario-math-reasoning-with-code-interpreter","slug":"mario-math-reasoning-with-code-interpreter","title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","date":"2024-01-16","arxiv_id":"2401.08190","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mario-math-reasoning/mario"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-preference-optimization-pushing","slug":"contrastive-preference-optimization-pushing","title":"Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation","date":"2024-01-16","arxiv_id":"2401.08417","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/alma"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/code-generation-with-alphacodium-from-prompt","slug":"code-generation-with-alphacodium-from-prompt","title":"Code Generation with AlphaCodium: From Prompt Engineering to Flow Engineering","date":"2024-01-16","arxiv_id":"2401.08500","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["codium-ai/alphacodium"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/emollms-a-series-of-emotional-large-language","slug":"emollms-a-series-of-emotional-large-language","title":"EmoLLMs: A Series of Emotional Large Language Models and Annotation Tools for Comprehensive Affective Analysis","date":"2024-01-16","arxiv_id":"2401.08508","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":0,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lzw108/emollms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-johnny-can-persuade-llms-to-jailbreak","slug":"how-johnny-can-persuade-llms-to-jailbreak","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","date":"2024-01-12","arxiv_id":"2401.06373","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chats-lab/persuasive_jailbreaker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/few-shot-detection-of-machine-generated-text","slug":"few-shot-detection-of-machine-generated-text","title":"Few-Shot Detection of Machine-Generated Text using Style Representations","date":"2024-01-12","arxiv_id":"2401.06712","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["llnl/luar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/health-llm-large-language-models-for-health","slug":"health-llm-large-language-models-for-health","title":"Health-LLM: Large Language Models for Health Prediction via Wearable Sensor Data","date":"2024-01-12","arxiv_id":"2401.06866","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mitmedialab/health-llm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/debugbench-evaluating-debugging-capability-of","slug":"debugbench-evaluating-debugging-capability-of","title":"DebugBench: Evaluating Debugging Capability of Large Language Models","date":"2024-01-09","arxiv_id":"2401.04621","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["thunlp/debugbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/marg-multi-agent-review-generation-for","slug":"marg-multi-agent-review-generation-for","title":"MARG: Multi-Agent Review Generation for Scientific Papers","date":"2024-01-08","arxiv_id":"2401.04259","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/marg-reviewer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/escalation-risks-from-language-models-in","slug":"escalation-risks-from-language-models-in","title":"Escalation Risks from Language Models in Military and Diplomatic Decision-Making","date":"2024-01-07","arxiv_id":"2401.03408","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jprivera44/EscalAItion"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/infobench-evaluating-instruction-following","slug":"infobench-evaluating-instruction-following","title":"InFoBench: Evaluating Instruction Following Ability in Large Language Models","date":"2024-01-07","arxiv_id":"2401.03601","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qinyiwei/infobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"286350930156f1fa491067f2e8c700eec7f1525e270cf72479c906470a5119c8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}