{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/ran/4","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":4,"pages_in_order":6,"rows_per_page":100,"rows":[301,400],"of":526,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4/papers/ran/1","prev":"/method/gpt-4/papers/ran/3","next":"/method/gpt-4/papers/ran/5","papers":[{"paper":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/aigcbench-comprehensive-evaluation-of-image","slug":"aigcbench-comprehensive-evaluation-of-image","title":"AIGCBench: Comprehensive Evaluation of Image-to-Video Content Generated by AI","date":"2024-01-03","arxiv_id":"2401.01651","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["benchcouncil/aigcbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-play-fine-tuning-converts-weak-language","slug":"self-play-fine-tuning-converts-weak-language","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","date":"2024-01-02","arxiv_id":"2401.01335","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uclaml/SPIN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-b-b-a-triggering-logical-reasoning-failures","slug":"a-b-b-a-triggering-logical-reasoning-failures","title":"LogicAsker: Evaluating and Improving the Logical Reasoning Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.00757","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yxwan123/logicasker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ragtruth-a-hallucination-corpus-for","slug":"ragtruth-a-hallucination-corpus-for","title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","date":"2023-12-31","arxiv_id":"2401.00396","n_code_links":3,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["particlemedia/ragtruth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","n_code_links":2,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/mac-sql-multi-agent-collaboration-for-text-to","slug":"mac-sql-multi-agent-collaboration-for-text-to","title":"MAC-SQL: A Multi-Agent Collaborative Framework for Text-to-SQL","date":"2023-12-18","arxiv_id":"2312.11242","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wbbeyourself/mac-sql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/nomiracl-knowing-when-you-don-t-know-for","slug":"nomiracl-knowing-when-you-don-t-know-for","title":"\"Knowing When You Don't Know\": A Multilingual Relevance Assessment Dataset for Robust Retrieval-Augmented Generation","date":"2023-12-18","arxiv_id":"2312.11361","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["project-miracl/nomiracl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/holodeck-language-guided-generation-of-3d","slug":"holodeck-language-guided-generation-of-3d","title":"Holodeck: Language Guided Generation of 3D Embodied AI Environments","date":"2023-12-14","arxiv_id":"2312.09067","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/Holodeck"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chat-3d-v2-bridging-3d-scene-and-large","slug":"chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","arxiv_id":"2312.08168","n_code_links":2,"syntology":{"ran":11,"of":11,"n_ran_checked":6,"n_instrument":5,"unverified":0,"pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chat-3d/chat-3d-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned","slug":"safety-alignment-in-nlp-tasks-weakly-aligned","title":"Safety Alignment in NLP Tasks: Weakly Aligned Summarization as an In-Context Attack","date":"2023-12-12","arxiv_id":"2312.06924","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fyyfu/safetyalignnlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ai-control-improving-safety-despite","slug":"ai-control-improving-safety-despite","title":"AI Control: Improving Safety Despite Intentional Subversion","date":"2023-12-12","arxiv_id":"2312.06942","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rgreenblatt/control-evaluations"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":6,"n_instrument":3,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/kwaiagents-generalized-information-seeking","slug":"kwaiagents-generalized-information-seeking","title":"KwaiAgents: Generalized Information-seeking Agent System with Large Language Models","date":"2023-12-08","arxiv_id":"2312.04889","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwaikeg/kwaiagents"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-4v-with-emotion-a-zero-shot-benchmark-for","slug":"gpt-4v-with-emotion-a-zero-shot-benchmark-for","title":"GPT-4V with Emotion: A Zero-shot Benchmark for Generalized Emotion Recognition","date":"2023-12-07","arxiv_id":"2312.04293","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zeroqiaoba/gpt4v-emotion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fortify-the-shortest-stave-in-attention","slug":"fortify-the-shortest-stave-in-attention","title":"Fortify the Shortest Stave in Attention: Enhancing Context Awareness of Large Language Models for Effective Tool Use","date":"2023-12-07","arxiv_id":"2312.04455","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fiorina1212/attention-buckets"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/quilt-llava-visual-instruction-tuning-by","slug":"quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","arxiv_id":"2312.04746","n_code_links":2,"syntology":{"ran":11,"of":12,"n_ran_checked":8,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/instructta-instruction-tuned-targeted-attack","slug":"instructta-instruction-tuned-targeted-attack","title":"InstructTA: Instruction-Tuned Targeted Attack for Large Vision-Language Models","date":"2023-12-04","arxiv_id":"2312.01886","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":6,"n_instrument":2,"unverified":4,"pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["xunguangwang/instructta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/tree-of-attacks-jailbreaking-black-box-llms","slug":"tree-of-attacks-jailbreaking-black-box-llms","title":"Tree of Attacks: Jailbreaking Black-Box LLMs Automatically","date":"2023-12-04","arxiv_id":"2312.02119","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ricommunity/tap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/d-bot-database-diagnosis-system-using-large","slug":"d-bot-database-diagnosis-system-using-large","title":"D-Bot: Database Diagnosis System using Large Language Models","date":"2023-12-03","arxiv_id":"2312.01454","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tsinghuadatabasegroup/db-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/critiquellm-scaling-llm-as-critic-for","slug":"critiquellm-scaling-llm-as-critic-for","title":"CritiqueLLM: Towards an Informative Critique Generation Model for Evaluation of Large Language Model Generation","date":"2023-11-30","arxiv_id":"2311.18702","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thu-coai/critiquellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unnatural-error-correction-gpt-4-can-almost","slug":"unnatural-error-correction-gpt-4-can-almost","title":"Unnatural Error Correction: GPT-4 Can Almost Perfectly Handle Unnatural Scrambled Text","date":"2023-11-30","arxiv_id":"2311.18805","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ccqq77/unnatural-error-correction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-generalist-foundation-models-outcompete","slug":"can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","arxiv_id":"2311.16452","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/gpt4vis-what-can-gpt-4-do-for-zero-shot","slug":"gpt4vis-what-can-gpt-4-do-for-zero-shot","title":"GPT4Vis: What Can GPT-4 Do for Zero-shot Visual Recognition?","date":"2023-11-27","arxiv_id":"2311.15732","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whwu95/GPT4Vis"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/meditron-70b-scaling-medical-pretraining-for","slug":"meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","arxiv_id":"2311.16079","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":9,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["epfllm/meditron"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-improving-document-understanding-an","slug":"towards-improving-document-understanding-an","title":"Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs","date":"2023-11-22","arxiv_id":"2311.13194","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["harrytea/tgdoc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-prompting-for-agi-systems","slug":"meta-prompting-for-agi-systems","title":"Meta Prompting for AI Systems","date":"2023-11-20","arxiv_id":"2311.11482","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["meta-prompting/meta-prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evil-geniuses-delving-into-the-safety-of-llm","slug":"evil-geniuses-delving-into-the-safety-of-llm","title":"Evil Geniuses: Delving into the Safety of LLM-based Agents","date":"2023-11-20","arxiv_id":"2311.11855","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["t1ans1r/evil-geniuses"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpqa-a-graduate-level-google-proof-q-a","slug":"gpqa-a-graduate-level-google-proof-q-a","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","date":"2023-11-20","arxiv_id":"2311.12022","n_code_links":3,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["idavidrein/gpqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/score-a-framework-for-self-contradictory","slug":"score-a-framework-for-self-contradictory","title":"Self-Contradictory Reasoning Evaluation and Detection","date":"2023-11-16","arxiv_id":"2311.09603","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["uscnlp-lime/Self-Contradictory"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/structured-chemistry-reasoning-with-large","slug":"structured-chemistry-reasoning-with-large","title":"Structured Chemistry Reasoning with Large Language Models","date":"2023-11-16","arxiv_id":"2311.09656","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ozyyshr/structchem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/huatuogpt-ii-one-stage-training-for-medical","slug":"huatuogpt-ii-one-stage-training-for-medical","title":"HuatuoGPT-II, One-stage Training for Medical Adaption of LLMs","date":"2023-11-16","arxiv_id":"2311.09774","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":11,"n_instrument":2,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["freedomintelligence/huatuogpt-ii"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/knowledgemath-knowledge-intensive-math-word","slug":"knowledgemath-knowledge-intensive-math-word","title":"FinanceMath: Knowledge-Intensive Math Reasoning in Finance Domains","date":"2023-11-16","arxiv_id":"2311.09797","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yale-nlp/knowledgemath"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/ml-bench-large-language-models-leverage-open","slug":"ml-bench-large-language-models-leverage-open","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","date":"2023-11-16","arxiv_id":"2311.09835","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gersteinlab/ml-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/factcheck-gpt-end-to-end-fine-grained","slug":"factcheck-gpt-end-to-end-fine-grained","title":"Factcheck-Bench: Fine-Grained Evaluation Benchmark for Automatic Fact-checkers","date":"2023-11-15","arxiv_id":"2311.09000","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuxiaw/factcheck-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tooltalk-evaluating-tool-usage-in-a","slug":"tooltalk-evaluating-tool-usage-in-a","title":"ToolTalk: Evaluating Tool-Usage in a Conversational Setting","date":"2023-11-15","arxiv_id":"2311.10775","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/a-wolf-in-sheep-s-clothing-generalized-nested","slug":"a-wolf-in-sheep-s-clothing-generalized-nested","title":"A Wolf in Sheep's Clothing: Generalized Nested Jailbreak Prompts can Fool Large Language Models Easily","date":"2023-11-14","arxiv_id":"2311.08268","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["NJUNLP/ReNeLLM"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/magic-benchmarking-large-language-model","slug":"magic-benchmarking-large-language-model","title":"MAgIC: Investigation of Large Language Model Powered Multi-Agent in Cognition, Adaptability, Rationality and Collaboration","date":"2023-11-14","arxiv_id":"2311.08562","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cathyxl/magic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/veritymath-advancing-mathematical-reasoning","slug":"veritymath-advancing-mathematical-reasoning","title":"VerityMath: Advancing Mathematical Reasoning by Self-Verification Through Unit Consistency","date":"2023-11-13","arxiv_id":"2311.07172","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vernontoh/veritymath"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/assessing-logical-puzzle-solving-in-large","slug":"assessing-logical-puzzle-solving-in-large","title":"Assessing Logical Puzzle Solving in Large Language Models: Insights from a Minesweeper Case Study","date":"2023-11-13","arxiv_id":"2311.07387","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yinghao-li/minesweeper-for-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-contamination-quiz-a-tool-to-detect-and","slug":"data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shahriargolchin/dcq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/technical-report-large-language-models-can","slug":"technical-report-large-language-models-can","title":"Large Language Models can Strategically Deceive their Users when Put Under Pressure","date":"2023-11-09","arxiv_id":"2311.07590","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apolloresearch/insider-trading"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/black-box-prompt-optimization-aligning-large","slug":"black-box-prompt-optimization-aligning-large","title":"Black-Box Prompt Optimization: Aligning Large Language Models without Model Training","date":"2023-11-07","arxiv_id":"2311.04155","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-coai/bpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/deepinception-hypnotize-large-language-model","slug":"deepinception-hypnotize-large-language-model","title":"DeepInception: Hypnotize Large Language Model to Be Jailbreaker","date":"2023-11-06","arxiv_id":"2311.03191","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tmlr-group/deepinception"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-llms-follow-simple-rules","slug":"can-llms-follow-simple-rules","title":"Can LLMs Follow Simple Rules?","date":"2023-11-06","arxiv_id":"2311.04235","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["normster/llm_rules"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/pptc-benchmark-evaluating-large-language","slug":"pptc-benchmark-evaluating-large-language","title":"PPTC Benchmark: Evaluating Large Language Models for PowerPoint Task Completion","date":"2023-11-03","arxiv_id":"2311.01767","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":13,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gydpku/pptc"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/soul-towards-sentiment-and-opinion","slug":"soul-towards-sentiment-and-opinion","title":"SOUL: Towards Sentiment and Opinion Understanding of Language","date":"2023-10-27","arxiv_id":"2310.17924","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["damo-nlp-sg/soul"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-for-aspect-based","slug":"large-language-models-for-aspect-based","title":"Large language models for aspect-based sentiment analysis","date":"2023-10-27","arxiv_id":"2310.18025","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qagentur/absa_llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/competeai-understanding-the-competition","slug":"competeai-understanding-the-competition","title":"CompeteAI: Understanding the Competition Dynamics in Large Language Model-based Agents","date":"2023-10-26","arxiv_id":"2310.17512","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["microsoft/competeai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-performance-predictors-are-good","slug":"llm-performance-predictors-are-good","title":"LLM Performance Predictors are good initializers for Architecture Search","date":"2023-10-25","arxiv_id":"2310.16712","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ubc-nlp/llmas"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/musr-testing-the-limits-of-chain-of-thought","slug":"musr-testing-the-limits-of-chain-of-thought","title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","date":"2023-10-24","arxiv_id":"2310.16049","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zayne-sprague/musr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-spatial-understanding-of-large","slug":"evaluating-spatial-understanding-of-large","title":"Evaluating Spatial Understanding of Large Language Models","date":"2023-10-23","arxiv_id":"2310.14540","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["runopti/spatialevalllm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/alpacare-instruction-tuned-large-language","slug":"alpacare-instruction-tuned-large-language","title":"AlpaCare:Instruction-tuned Large Language Models for Medical Application","date":"2023-10-23","arxiv_id":"2310.14558","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xzhang97666/alpacare"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/linc-a-neurosymbolic-approach-for-logical","slug":"linc-a-neurosymbolic-approach-for-logical","title":"LINC: A Neurosymbolic Approach for Logical Reasoning by Combining Language Models with First-Order Logic Provers","date":"2023-10-23","arxiv_id":"2310.15164","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":0,"n_instrument":1,"unverified":6,"pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["benlipkin/linc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/cxr-llava-multimodal-large-language-model-for","slug":"cxr-llava-multimodal-large-language-model-for","title":"CXR-LLAVA: a multimodal large language model for interpreting chest X-ray images","date":"2023-10-22","arxiv_id":"2310.18341","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ecofri/cxr_llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/gemba-mqm-detecting-translation-quality-error","slug":"gemba-mqm-detecting-translation-quality-error","title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","date":"2023-10-21","arxiv_id":"2310.13988","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/small-language-models-fine-tuned-to","slug":"small-language-models-fine-tuned-to","title":"Small Language Models Fine-tuned to Coordinate Larger Language Models improve Complex Reasoning","date":"2023-10-21","arxiv_id":"2310.18338","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lcs2-iiitd/daslam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cache-me-if-you-can-an-online-cost-aware","slug":"cache-me-if-you-can-an-online-cost-aware","title":"Cache me if you Can: an Online Cost-aware Teacher-Student framework to Reduce the Calls to Large Language Models","date":"2023-10-20","arxiv_id":"2310.13395","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stoyian/OCaTS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/botchat-evaluating-llms-capabilities-of","slug":"botchat-evaluating-llms-capabilities-of","title":"BotChat: Evaluating LLMs' Capabilities of Having Multi-Turn Dialogues","date":"2023-10-20","arxiv_id":"2310.13650","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["open-compass/botchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluation-metrics-in-the-era-of-gpt-4","slug":"evaluation-metrics-in-the-era-of-gpt-4","title":"Evaluation Metrics in the Era of GPT-4: Reliably Evaluating Large Language Models on Sequence to Sequence Tasks","date":"2023-10-20","arxiv_id":"2310.13800","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["protagolabs/seq2seq_llm_evaluation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/agenttuning-enabling-generalized-agent","slug":"agenttuning-enabling-generalized-agent","title":"AgentTuning: Enabling Generalized Agent Abilities for LLMs","date":"2023-10-19","arxiv_id":"2310.12823","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thudm/agenttuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":9,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/automix-automatically-mixing-language-models","slug":"automix-automatically-mixing-language-models","title":"AutoMix: Automatically Mixing Language Models","date":"2023-10-19","arxiv_id":"2310.12963","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["automix-llm/automix"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/probing-the-creativity-of-large-language","slug":"probing-the-creativity-of-large-language","title":"Probing the Creativity of Large Language Models: Can models produce divergent semantic association?","date":"2023-10-17","arxiv_id":"2310.11158","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dingnlab/probing_creativity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/compost-characterizing-and-evaluating","slug":"compost-characterizing-and-evaluating","title":"CoMPosT: Characterizing and Evaluating Caricature in LLM Simulations","date":"2023-10-17","arxiv_id":"2310.11501","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["myracheng/lm_caricature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/trigo-benchmarking-formal-mathematical-proof","slug":"trigo-benchmarking-formal-mathematical-proof","title":"TRIGO: Benchmarking Formal Mathematical Proof Reduction for Generative Language Models","date":"2023-10-16","arxiv_id":"2310.10180","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["menik1126/TRIGO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/bioplanner-automatic-evaluation-of-llms-on","slug":"bioplanner-automatic-evaluation-of-llms-on","title":"BioPlanner: Automatic Evaluation of LLMs on Protocol Planning in Biology","date":"2023-10-16","arxiv_id":"2310.10632","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bioplanner/bioplanner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dont-add-dont-miss-effective-content","slug":"dont-add-dont-miss-effective-content","title":"Dont Add, dont Miss: Effective Content Preserving Generation from Pre-Selected Text Spans","date":"2023-10-13","arxiv_id":"2310.09017","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lovodkin93/cdr_ctr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-systematic-evaluation-of-large-language-1","slug":"a-systematic-evaluation-of-large-language-1","title":"Assessing and Enhancing the Robustness of Large Language Models with Task Structure Variations for Logical Reasoning","date":"2023-10-13","arxiv_id":"2310.09430","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["strong-ai-lab/logical-and-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpreting-reward-models-in-rlhf-tuned","slug":"interpreting-reward-models-in-rlhf-tuned","title":"Interpreting Learned Feedback Patterns in Large Language Models","date":"2023-10-12","arxiv_id":"2310.08164","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["apartresearch/interpreting-reward-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/prometheus-inducing-fine-grained-evaluation","slug":"prometheus-inducing-fine-grained-evaluation","title":"Prometheus: Inducing Fine-grained Evaluation Capability in Language Models","date":"2023-10-12","arxiv_id":"2310.08491","n_code_links":3,"syntology":{"ran":4,"of":9,"n_ran_checked":3,"n_instrument":1,"unverified":5,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["kaistAI/Prometheus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","n_code_links":1,"syntology":{"ran":12,"of":12,"n_ran_checked":6,"n_instrument":6,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-are-zero-shot-time-1","slug":"large-language-models-are-zero-shot-time-1","title":"Large Language Models Are Zero-Shot Time Series Forecasters","date":"2023-10-11","arxiv_id":"2310.07820","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ngruver/llmtime"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/swe-bench-can-language-models-resolve-real","slug":"swe-bench-can-language-models-resolve-real","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","date":"2023-10-10","arxiv_id":"2310.06770","n_code_links":8,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/put-your-money-where-your-mouth-is-evaluating","slug":"put-your-money-where-your-mouth-is-evaluating","title":"Put Your Money Where Your Mouth Is: Evaluating Strategic Planning and Execution of LLM Agents in an Auction Arena","date":"2023-10-09","arxiv_id":"2310.05746","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jiangjiechen/auction-arena"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/reinforcement-learning-in-the-era-of-llms","slug":"reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","arxiv_id":"2310.06147","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/harnessing-the-power-of-large-language-models-1","slug":"harnessing-the-power-of-large-language-models-1","title":"Harnessing the Power of Large Language Models for Empathetic Response Generation: Empirical Investigations and Improvements","date":"2023-10-08","arxiv_id":"2310.05140","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["27182812/LLM4ED"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-large-language-models-be-good-path","slug":"can-large-language-models-be-good-path","title":"Can Large Language Models be Good Path Planners? A Benchmark and Investigation on Spatial-temporal Reasoning","date":"2023-10-05","arxiv_id":"2310.03249","n_code_links":1,"syntology":{"ran":28,"of":28,"n_ran_checked":28,"n_instrument":0,"unverified":0,"pointer_only":28,"phrase":"28 ran (of which 0 constructed an object rather than computing a result; 28 with no instrument failure: 0 honoured, 0 violated, 28 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mohamedaghzal/llms-as-path-planners"],"state":"official (archive's flag): 28 ran","n_ran":28,"n_constructed":0,"n_ran_no_instrument_failure":28,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","n_code_links":2,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-hallucinations-in-chinese-large","slug":"evaluating-hallucinations-in-chinese-large","title":"Evaluating Hallucinations in Chinese Large Language Models","date":"2023-10-05","arxiv_id":"2310.03368","n_code_links":3,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xiami2019/halluqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mathcoder-seamless-code-integration-in-llms","slug":"mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","arxiv_id":"2310.03731","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/automating-human-tutor-style-programming","slug":"automating-human-tutor-style-programming","title":"Automating Human Tutor-Style Programming Feedback: Leveraging GPT-4 Tutor Model for Hint Generation and GPT-3.5 Student Model for Hint Validation","date":"2023-10-05","arxiv_id":"2310.03780","n_code_links":2,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["machine-teaching-group/lak2024_gpt4-hints-gpt3.5val","machine-teaching-group/lak2024_gpt4hints-gpt3.5val"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dq-lore-dual-queries-with-low-rank","slug":"dq-lore-dual-queries-with-low-rank","title":"DQ-LoRe: Dual Queries with Low Rank Approximation Re-ranking for In-Context Learning","date":"2023-10-04","arxiv_id":"2310.02954","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ai4fun/dq-lore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/t-3-bench-benchmarking-current-progress-in","slug":"t-3-bench-benchmarking-current-progress-in","title":"T$^3$Bench: Benchmarking Current Progress in Text-to-3D Generation","date":"2023-10-04","arxiv_id":"2310.02977","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THU-LYJ-Lab/T3Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/halle-switch-rethinking-and-controlling","slug":"halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","arxiv_id":"2310.01779","n_code_links":2,"syntology":{"ran":6,"of":9,"n_ran_checked":3,"n_instrument":3,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bronyayang/HallE_Switch","bronyayang/halle_control"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-taught-optimizer-stop-recursively-self","slug":"self-taught-optimizer-stop-recursively-self","title":"Self-Taught Optimizer (STOP): Recursively Self-Improving Code Generation","date":"2023-10-03","arxiv_id":"2310.02304","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/stop"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/ecoassistant-using-llm-assistant-more","slug":"ecoassistant-using-llm-assistant-more","title":"EcoAssistant: Using LLM Assistant More Affordably and Accurately","date":"2023-10-03","arxiv_id":"2310.03046","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jieyuz2/ecoassistant"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ultrafeedback-boosting-language-models-with","slug":"ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","arxiv_id":"2310.01377","n_code_links":4,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thunlp/ultrafeedback"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-entity-deduction-arena-a-playground-for","slug":"the-entity-deduction-arena-a-playground-for","title":"Probing the Multi-turn Planning Capabilities of LLMs via 20 Question Games","date":"2023-10-02","arxiv_id":"2310.01468","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["apple/ml-entity-deduction-arena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tigerscore-towards-building-explainable","slug":"tigerscore-towards-building-explainable","title":"TIGERScore: Towards Building Explainable Metric for All Text Generation Tasks","date":"2023-10-01","arxiv_id":"2310.00752","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["TIGER-AI-Lab/TIGERScore"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/booookscore-a-systematic-exploration-of-book","slug":"booookscore-a-systematic-exploration-of-book","title":"BooookScore: A systematic exploration of book-length summarization in the era of LLMs","date":"2023-10-01","arxiv_id":"2310.00785","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lilakk/booookscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/automatikz-text-guided-synthesis-of","slug":"automatikz-text-guided-synthesis-of","title":"AutomaTikZ: Text-Guided Synthesis of Scientific Vector Graphics with TikZ","date":"2023-09-30","arxiv_id":"2310.00367","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["potamides/automatikz"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-deliberation-evaluating-llms-with","slug":"llm-deliberation-evaluating-llms-with","title":"Cooperation, Competition, and Maliciousness: LLM-Stakeholders Interactive Negotiation","date":"2023-09-29","arxiv_id":"2309.17234","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["s-abdelnabi/llm-deliberation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-large-language-models-in-coding","slug":"enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","arxiv_id":"2309.17272","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":7,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["skpig/MPSC"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}}],"record_sha256":"0e3f2d3c1526fc5b9d5552e0f66dddebd02dfde85ab3a1f0283c451615af7944","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}