{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/15","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":15,"pages_in_order":29,"rows_per_page":100,"rows":[1401,1500],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/14","next":"/method/gpt-4/papers/16","papers":[{"paper":"/paper/ai-enhanced-cognitive-behavioral-therapy-deep","slug":"ai-enhanced-cognitive-behavioral-therapy-deep","title":"AI-Enhanced Cognitive Behavioral Therapy: Deep Learning and Large Language Models for Extracting Cognitive Pathways from Social Media Texts","date":"2024-04-17","arxiv_id":"2404.11449","n_code_links":1,"syntology":null},{"paper":null,"slug":"octopus-v3-technical-report-for-on-device-sub","title":"Octopus v3: Technical Report for On-device Sub-billion Multimodal AI Agent","date":"2024-04-17","arxiv_id":"2404.11459","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-optimizer-of-text-to-image-diffusion","title":"Prompt Optimizer of Text-to-Image Diffusion Models for Abstract Concept Understanding","date":"2024-04-17","arxiv_id":"2404.11589","n_code_links":0,"syntology":null},{"paper":"/paper/rd2bench-toward-data-centric-automatic-r-d","slug":"rd2bench-toward-data-centric-automatic-r-d","title":"Towards Data-Centric Automatic R&D","date":"2024-04-17","arxiv_id":"2404.11276","n_code_links":1,"syntology":null},{"paper":"/paper/can-language-models-solve-olympiad","slug":"can-language-models-solve-olympiad","title":"Can Language Models Solve Olympiad Programming?","date":"2024-04-16","arxiv_id":"2404.10952","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["princeton-nlp/USACO"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"cotar-chain-of-thought-attribution-reasoning","title":"CoTAR: Chain-of-Thought Attribution Reasoning with Multi-level Granularity","date":"2024-04-16","arxiv_id":"2404.10513","n_code_links":0,"syntology":null},{"paper":"/paper/how-faithful-are-rag-models-quantifying-the","slug":"how-faithful-are-rag-models-quantifying-the","title":"ClashEval: Quantifying the tug-of-war between an LLM's internal prior and external evidence","date":"2024-04-16","arxiv_id":"2404.10198","n_code_links":1,"syntology":null},{"paper":"/paper/incubating-text-classifiers-following-user","slug":"incubating-text-classifiers-following-user","title":"Incubating Text Classifiers Following User Instruction with Nothing but LLM","date":"2024-04-16","arxiv_id":"2404.10877","n_code_links":1,"syntology":null},{"paper":"/paper/minicheck-efficient-fact-checking-of-llms-on","slug":"minicheck-efficient-fact-checking-of-llms-on","title":"MiniCheck: Efficient Fact-Checking of LLMs on Grounding Documents","date":"2024-04-16","arxiv_id":"2404.10774","n_code_links":2,"syntology":{"ran":2,"of":8,"n_ran_checked":2,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["liyan06/minicheck"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/search-beyond-queries-training-smaller","slug":"search-beyond-queries-training-smaller","title":"Grounded Language Agent for Product Search via Intelligent Web Interactions","date":"2024-04-16","arxiv_id":"2404.10887","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MultifacetedNLP/Web-Agents-Unsupervised"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-supervised-visual-preference-alignment","slug":"self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","arxiv_id":"2404.10501","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":6,"n_instrument":4,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Kevinz-code/SeVa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"social-choice-for-ai-alignment-dealing-with","title":"Social Choice Should Guide AI Alignment in Dealing with Diverse Human Feedback","date":"2024-04-16","arxiv_id":"2404.10271","n_code_links":0,"syntology":null},{"paper":null,"slug":"learn-your-reference-model-for-real-good","title":"Learn Your Reference Model for Real Good Alignment","date":"2024-04-15","arxiv_id":"2404.09656","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-evaluators-recognize-and-favor-their-own","title":"LLM Evaluators Recognize and Favor Their Own Generations","date":"2024-04-15","arxiv_id":"2404.13076","n_code_links":0,"syntology":null},{"paper":null,"slug":"numerical-attributes-learning-for-cardiac","title":"Are Medium-Sized Transformers Models still Relevant for Medical Records Processing?","date":"2024-04-15","arxiv_id":"2404.10171","n_code_links":0,"syntology":null},{"paper":null,"slug":"unveiling-imitation-learning-exploring-the","title":"Unveiling Imitation Learning: Exploring the Impact of Data Falsity to Large Language Model","date":"2024-04-15","arxiv_id":"2404.09717","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-building-age-classification-from","slug":"zero-shot-building-age-classification-from","title":"Zero-shot Building Age Classification from Facade Image Using GPT-4","date":"2024-04-15","arxiv_id":"2404.09921","n_code_links":1,"syntology":null},{"paper":"/paper/constrained-c-test-generation-via-mixed","slug":"constrained-c-test-generation-via-mixed","title":"Constrained C-Test Generation via Mixed-Integer Programming","date":"2024-04-12","arxiv_id":"2404.08821","n_code_links":1,"syntology":null},{"paper":"/paper/dataset-reset-policy-optimization-for-rlhf","slug":"dataset-reset-policy-optimization-for-rlhf","title":"Dataset Reset Policy Optimization for RLHF","date":"2024-04-12","arxiv_id":"2404.08495","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cornell-rl/drpo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"don-t-forget-to-put-the-milk-back-dataset-for","title":"\"Don't forget to put the milk back!\" Dataset for Enabling Embodied Agents to Detect Anomalous Situations","date":"2024-04-12","arxiv_id":"2404.08827","n_code_links":0,"syntology":null},{"paper":"/paper/small-models-are-still-effective-cross-domain","slug":"small-models-are-still-effective-cross-domain","title":"Small Models Are (Still) Effective Cross-Domain Argument Extractors","date":"2024-04-12","arxiv_id":"2404.08579","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-generation-and-evaluation-of","slug":"automatic-generation-and-evaluation-of","title":"Automatic Generation and Evaluation of Reading Comprehension Test Items with Large Language Models","date":"2024-04-11","arxiv_id":"2404.07720","n_code_links":2,"syntology":null},{"paper":"/paper/comments-as-natural-logic-pivots-improve-code","slug":"comments-as-natural-logic-pivots-improve-code","title":"Comments as Natural Logic Pivots: Improve Code Generation via Comment Perspective","date":"2024-04-11","arxiv_id":"2404.07549","n_code_links":1,"syntology":null},{"paper":"/paper/designqa-a-multimodal-benchmark-for","slug":"designqa-a-multimodal-benchmark-for","title":"DesignQA: A Multimodal Benchmark for Evaluating Large Language Models' Understanding of Engineering Documentation","date":"2024-04-11","arxiv_id":"2404.07917","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["anniedoris/design_qa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-words-to-numbers-your-large-language","slug":"from-words-to-numbers-your-large-language","title":"From Words to Numbers: Your Large Language Model Is Secretly A Capable Regressor When Given In-Context Examples","date":"2024-04-11","arxiv_id":"2404.07544","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["robertvacareanu/llm4regression"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-latency-conversational-turns-for-spoken","title":"Human Latency Conversational Turns for Spoken Avatar Systems","date":"2024-04-11","arxiv_id":"2404.16053","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-agents-can-autonomously-exploit-one-day","title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","date":"2024-04-11","arxiv_id":"2404.08144","n_code_links":0,"syntology":null},{"paper":null,"slug":"mm-phyqa-multimodal-physics-question","title":"MM-PhyQA: Multimodal Physics Question-Answering With Multi-Image CoT Prompting","date":"2024-04-11","arxiv_id":"2404.08704","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-generation-of-personalities-with","slug":"dynamic-generation-of-personalities-with","title":"Dynamic Generation of Personalities with Large Language Models","date":"2024-04-10","arxiv_id":"2404.07084","n_code_links":1,"syntology":null},{"paper":null,"slug":"characterizing-multimodal-long-form","title":"Characterizing Multimodal Long-form Summarization: A Case Study on Financial Reports","date":"2024-04-09","arxiv_id":"2404.06162","n_code_links":0,"syntology":null},{"paper":"/paper/llm2vec-large-language-models-are-secretly","slug":"llm2vec-large-language-models-are-secretly","title":"LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders","date":"2024-04-09","arxiv_id":"2404.05961","n_code_links":1,"syntology":null},{"paper":null,"slug":"llms-reading-comprehension-is-affected-by","title":"LLMs' Reading Comprehension Is Affected by Parametric Knowledge and Struggles with Hypothetical Statements","date":"2024-04-09","arxiv_id":"2404.06283","n_code_links":0,"syntology":null},{"paper":null,"slug":"sandwich-attack-multi-language-mixture","title":"Sandwich attack: Multi-language Mixture Adaptive Attack on LLMs","date":"2024-04-09","arxiv_id":"2404.07242","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-an-llm-in-identifying-logical","title":"Evaluation of an LLM in Identifying Logical Fallacies: A Call for Rigor When Adopting LLMs in HCI Research","date":"2024-04-08","arxiv_id":"2404.05213","n_code_links":0,"syntology":null},{"paper":"/paper/llm-reasoners-new-evaluation-library-and","slug":"llm-reasoners-new-evaluation-library-and","title":"LLM Reasoners: New Evaluation, Library, and Analysis of Step-by-Step Reasoning with Large Language Models","date":"2024-04-08","arxiv_id":"2404.05221","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"relation-extraction-using-large-language","title":"Relation Extraction Using Large Language Models: A Case Study on Acupuncture Point Locations","date":"2024-04-08","arxiv_id":"2404.05415","n_code_links":0,"syntology":null},{"paper":"/paper/use-of-a-structured-knowledge-base-enhances","slug":"use-of-a-structured-knowledge-base-enhances","title":"Use of a Structured Knowledge Base Enhances Metadata Curation by Large Language Models","date":"2024-04-08","arxiv_id":"2404.05893","n_code_links":1,"syntology":null},{"paper":"/paper/xiwu-a-basis-flexible-and-learnable-llm-for","slug":"xiwu-a-basis-flexible-and-learnable-llm-for","title":"Xiwu: A Basis Flexible and Learnable LLM for High Energy Physics","date":"2024-04-08","arxiv_id":"2404.08001","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-geometric-problem-solving-a","slug":"advancing-geometric-problem-solving-a","title":"MM-MATH: Advancing Multimodal Math Evaluation with Process Evaluation and Fine-grained Classification","date":"2024-04-07","arxiv_id":"2404.05091","n_code_links":1,"syntology":null},{"paper":null,"slug":"initial-exploration-of-zero-shot-privacy","title":"Initial Exploration of Zero-Shot Privacy Utility Tradeoffs in Tabular Data Using GPT-4","date":"2024-04-07","arxiv_id":"2404.05047","n_code_links":0,"syntology":null},{"paper":"/paper/macm-utilizing-a-multi-agent-system-for","slug":"macm-utilizing-a-multi-agent-system-for","title":"MACM: Utilizing a Multi-Agent System for Condition Mining in Solving Complex Mathematical Problems","date":"2024-04-06","arxiv_id":"2404.04735","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["bin123apple/macm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"cleared-for-takeoff-compositional-conditional","title":"Cleared for Takeoff? Compositional & Conditional Reasoning may be the Achilles Heel to (Flight-Booking) Language Agents","date":"2024-04-05","arxiv_id":"2404.04237","n_code_links":0,"syntology":null},{"paper":null,"slug":"effects-of-different-prompts-on-the-quality","title":"Effects of Different Prompts on the Quality of GPT-4 Responses to Dementia Care Questions","date":"2024-04-05","arxiv_id":"2404.08674","n_code_links":0,"syntology":null},{"paper":"/paper/scope-ambiguities-in-large-language-models","slug":"scope-ambiguities-in-large-language-models","title":"Scope Ambiguities in Large Language Models","date":"2024-04-05","arxiv_id":"2404.04332","n_code_links":1,"syntology":null},{"paper":"/paper/autowebglm-bootstrap-and-reinforce-a-large","slug":"autowebglm-bootstrap-and-reinforce-a-large","title":"AutoWebGLM: A Large Language Model-based Web Navigating Agent","date":"2024-04-04","arxiv_id":"2404.03648","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thudm/autowebglm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"capabilities-of-large-language-models-in","title":"Capabilities of Large Language Models in Control Engineering: A Benchmark Study on GPT-4, Claude 3 Opus, and Gemini 1.0 Ultra","date":"2024-04-04","arxiv_id":"2404.03647","n_code_links":0,"syntology":null},{"paper":"/paper/conversational-disease-diagnosis-via-external","slug":"conversational-disease-diagnosis-via-external","title":"Conversational Disease Diagnosis via External Planner-Controlled Large Language Models","date":"2024-04-04","arxiv_id":"2404.04292","n_code_links":1,"syntology":null},{"paper":null,"slug":"direct-nash-optimization-teaching-language","title":"Direct Nash Optimization: Teaching Language Models to Self-Improve with General Preferences","date":"2024-04-04","arxiv_id":"2404.03715","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-llms-at-detecting-errors-in-llm","slug":"evaluating-llms-at-detecting-errors-in-llm","title":"Evaluating LLMs at Detecting Errors in LLM Responses","date":"2024-04-04","arxiv_id":"2404.03602","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["psunlpgroup/realmistake"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/probing-large-language-models-for-scalar","slug":"probing-large-language-models-for-scalar","title":"Probing Large Language Models for Scalar Adjective Lexical Semantics and Scalar Diversity Pragmatics","date":"2024-04-04","arxiv_id":"2404.03301","n_code_links":1,"syntology":null},{"paper":null,"slug":"reason-from-fallacy-enhancing-large-language","title":"Reason from Fallacy: Enhancing Large Language Models' Logical Reasoning through Logical Fallacy Understanding","date":"2024-04-04","arxiv_id":"2404.04293","n_code_links":0,"syntology":null},{"paper":null,"slug":"attributions-toward-artificial-agents-in-a","title":"Attributions toward Artificial Agents in a modified Moral Turing Test","date":"2024-04-03","arxiv_id":"2406.11854","n_code_links":0,"syntology":null},{"paper":"/paper/bcamirs-at-semeval-2024-task-4-beyond-words-a","slug":"bcamirs-at-semeval-2024-task-4-beyond-words-a","title":"BCAmirs at SemEval-2024 Task 4: Beyond Words: A Multimodal and Multilingual Exploration of Persuasion in Memes","date":"2024-04-03","arxiv_id":"2404.03022","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-for-2","slug":"benchmarking-large-language-models-for-2","title":"Benchmarking Large Language Models for Persian: A Preliminary Study Focusing on ChatGPT","date":"2024-04-03","arxiv_id":"2404.02403","n_code_links":1,"syntology":null},{"paper":"/paper/conifer-improving-complex-constrained","slug":"conifer-improving-complex-constrained","title":"Conifer: Improving Complex Constrained Instruction-Following Ability of Large Language Models","date":"2024-04-03","arxiv_id":"2404.02823","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["coniferlm/conifer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"task-agnostic-architecture-for-algorithm","title":"Task Agnostic Architecture for Algorithm Induction via Implicit Composition","date":"2024-04-03","arxiv_id":"2404.02450","n_code_links":0,"syntology":null},{"paper":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be","slug":"utebc-nlp-at-semeval-2024-task-9-can-llms-be","title":"uTeBC-NLP at SemEval-2024 Task 9: Can LLMs be Lateral Thinkers?","date":"2024-04-03","arxiv_id":"2404.02474","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ipouyall/can-llms-be-lateral-thinkers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automated-user-story-generation-with-test","title":"Automated User Story Generation with Test Case Specification Using Large Language Model","date":"2024-04-02","arxiv_id":"2404.01558","n_code_links":0,"syntology":null},{"paper":null,"slug":"great-now-write-an-article-about-that-the","title":"Great, Now Write an Article About That: The Crescendo Multi-Turn LLM Jailbreak Attack","date":"2024-04-02","arxiv_id":"2404.01833","n_code_links":0,"syntology":null},{"paper":null,"slug":"indoculture-exploring-geographically","title":"IndoCulture: Exploring Geographically-Influenced Cultural Commonsense Reasoning Across Eleven Indonesian Provinces","date":"2024-04-02","arxiv_id":"2404.01854","n_code_links":0,"syntology":null},{"paper":"/paper/jailbreaking-leading-safety-aligned-llms-with","slug":"jailbreaking-leading-safety-aligned-llms-with","title":"Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks","date":"2024-04-02","arxiv_id":"2404.02151","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/llm-adaptive-attacks"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"metal-towards-multilingual-meta-evaluation","title":"METAL: Towards Multilingual Meta-Evaluation","date":"2024-04-02","arxiv_id":"2404.01667","n_code_links":0,"syntology":null},{"paper":null,"slug":"octopus-on-device-language-model-for-function","title":"Octopus: On-device language model for function calling of software APIs","date":"2024-04-02","arxiv_id":"2404.01549","n_code_links":0,"syntology":null},{"paper":null,"slug":"octopus-v2-on-device-language-model-for-super","title":"Octopus v2: On-device language model for super agent","date":"2024-04-02","arxiv_id":"2404.01744","n_code_links":0,"syntology":null},{"paper":"/paper/patch-psychometrics-assisted-benchmarking-of","slug":"patch-psychometrics-assisted-benchmarking-of","title":"PATCH! {P}sychometrics-{A}ssis{T}ed Ben{CH}marking of Large Language Models against Human Populations: A Case Study of Proficiency in 8th Grade Mathematics","date":"2024-04-02","arxiv_id":"2404.01799","n_code_links":1,"syntology":null},{"paper":"/paper/toward-informal-language-processing-knowledge","slug":"toward-informal-language-processing-knowledge","title":"Toward Informal Language Processing: Knowledge of Slang in Large Language Models","date":"2024-04-02","arxiv_id":"2404.02323","n_code_links":1,"syntology":null},{"paper":null,"slug":"automated-assessment-of-encouragement-and","title":"Automated Assessment of Encouragement and Warmth in Classrooms Leveraging Multimodal Emotional Features and ChatGPT","date":"2024-04-01","arxiv_id":"2404.15310","n_code_links":0,"syntology":null},{"paper":null,"slug":"forklift-an-extensible-neural-lifter","title":"Forklift: An Extensible Neural Lifter","date":"2024-04-01","arxiv_id":"2404.16041","n_code_links":0,"syntology":null},{"paper":null,"slug":"isobench-benchmarking-multimodal-foundation","title":"IsoBench: Benchmarking Multimodal Foundation Models on Isomorphic Representations","date":"2024-04-01","arxiv_id":"2404.01266","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-radjudge-achieving-radiologist-level","title":"LLM-RadJudge: Achieving Radiologist-Level Evaluation for X-Ray Report Generation","date":"2024-04-01","arxiv_id":"2404.00998","n_code_links":0,"syntology":null},{"paper":"/paper/unveiling-divergent-inductive-biases-of-llms","slug":"unveiling-divergent-inductive-biases-of-llms","title":"Unveiling Divergent Inductive Biases of LLMs on Temporal Data","date":"2024-04-01","arxiv_id":"2404.01453","n_code_links":1,"syntology":null},{"paper":null,"slug":"algorithmic-collusion-by-large-language","title":"Algorithmic Collusion by Large Language Models","date":"2024-03-31","arxiv_id":"2404.00806","n_code_links":0,"syntology":null},{"paper":"/paper/chops-chat-with-customer-profile-systems-for","slug":"chops-chat-with-customer-profile-systems-for","title":"CHOPS: CHat with custOmer Profile Systems for Customer Service with LLMs","date":"2024-03-31","arxiv_id":"2404.01343","n_code_links":1,"syntology":null},{"paper":"/paper/couda-coherence-evaluation-via-unified-data","slug":"couda-coherence-evaluation-via-unified-data","title":"CoUDA: Coherence Evaluation via Unified Data Augmentation","date":"2024-03-31","arxiv_id":"2404.00681","n_code_links":1,"syntology":null},{"paper":"/paper/evocodebench-an-evolving-code-generation","slug":"evocodebench-an-evolving-code-generation","title":"EvoCodeBench: An Evolving Code Generation Benchmark Aligned with Real-World Code Repositories","date":"2024-03-31","arxiv_id":"2404.00599","n_code_links":1,"syntology":{"ran":15,"of":15,"n_ran_checked":14,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 3 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["seketeam/evocodebench"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/extracting-social-determinants-of-health-from","slug":"extracting-social-determinants-of-health-from","title":"Extracting Social Determinants of Health from Pediatric Patient Notes Using Large Language Models: Novel Corpus and Methods","date":"2024-03-31","arxiv_id":"2404.00826","n_code_links":1,"syntology":null},{"paper":"/paper/how-much-are-llms-contaminated-a","slug":"how-much-are-llms-contaminated-a","title":"How Much are Large Language Models Contaminated? A Comprehensive Survey and the LLMSanitize Library","date":"2024-03-31","arxiv_id":"2404.00699","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ntunlp/llmsanitize"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-llms-master-math-investigating-large","slug":"can-llms-master-math-investigating-large","title":"Can LLMs Master Math? Investigating Large Language Models on Math Stack Exchange","date":"2024-03-30","arxiv_id":"2404.00344","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["gipplab/llm-investig-mathstackexchange"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/edinburgh-clinical-nlp-at-semeval-2024-task-2","slug":"edinburgh-clinical-nlp-at-semeval-2024-task-2","title":"Edinburgh Clinical NLP at SemEval-2024 Task 2: Fine-tune your model unless you have access to GPT-4","date":"2024-03-30","arxiv_id":"2404.00484","n_code_links":1,"syntology":null},{"paper":null,"slug":"injecting-new-knowledge-into-large-language","title":"Injecting New Knowledge into Large Language Models via Supervised Fine-Tuning","date":"2024-03-30","arxiv_id":"2404.00213","n_code_links":0,"syntology":null},{"paper":"/paper/small-language-models-learn-enhanced","slug":"small-language-models-learn-enhanced","title":"Small Language Models Learn Enhanced Reasoning Skills from Medical Textbooks","date":"2024-03-30","arxiv_id":"2404.00376","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-the-general-agent-capabilities-of","slug":"enhancing-the-general-agent-capabilities-of","title":"Enhancing the General Agent Capabilities of Low-Parameter LLMs through Tuning and Multi-Branch Reasoning","date":"2024-03-29","arxiv_id":"2403.19962","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["haiv-lab/llm-tmbr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/latxa-an-open-language-model-and-evaluation","slug":"latxa-an-open-language-model-and-evaluation","title":"Latxa: An Open Language Model and Evaluation Suite for Basque","date":"2024-03-29","arxiv_id":"2403.20266","n_code_links":1,"syntology":null},{"paper":"/paper/mango-a-benchmark-for-evaluating-mapping-and","slug":"mango-a-benchmark-for-evaluating-mapping-and","title":"MANGO: A Benchmark for Evaluating Mapping and Navigation Abilities of Large Language Models","date":"2024-03-29","arxiv_id":"2403.19913","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oaklight/mango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-fly-definition-augmentation-of-llms","slug":"on-the-fly-definition-augmentation-of-llms","title":"On-the-fly Definition Augmentation of LLMs for Biomedical NER","date":"2024-03-29","arxiv_id":"2404.00152","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/beacon"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"realm-reference-resolution-as-language","title":"ReALM: Reference Resolution As Language Modeling","date":"2024-03-29","arxiv_id":"2403.20329","n_code_links":0,"syntology":null},{"paper":null,"slug":"checkpoint-merging-via-bayesian-optimization","title":"Checkpoint Merging via Bayesian Optimization in LLM Pretraining","date":"2024-03-28","arxiv_id":"2403.19390","n_code_links":0,"syntology":null},{"paper":null,"slug":"generate-then-retrieve-conversational","title":"Generating Multi-Aspect Queries for Conversational Search","date":"2024-03-28","arxiv_id":"2403.19302","n_code_links":0,"syntology":null},{"paper":"/paper/mateval-a-multi-agent-discussion-framework","slug":"mateval-a-multi-agent-discussion-framework","title":"MATEval: A Multi-Agent Discussion Framework for Advancing Open-Ended Text Evaluation","date":"2024-03-28","arxiv_id":"2403.19305","n_code_links":1,"syntology":null},{"paper":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"blade-enhancing-black-box-large-language","title":"BLADE: Enhancing Black-box Large Language Models with Small Domain-Specific Models","date":"2024-03-27","arxiv_id":"2403.18365","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-large-language-models-for-health-1","title":"Evaluating Large Language Models for Health-Related Text Classification Tasks with Public Social Media Data","date":"2024-03-27","arxiv_id":"2403.19031","n_code_links":0,"syntology":null},{"paper":"/paper/long-form-factuality-in-large-language-models","slug":"long-form-factuality-in-large-language-models","title":"Long-form factuality in large language models","date":"2024-03-27","arxiv_id":"2403.18802","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-deepmind/long-form-factuality"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/mini-gemini-mining-the-potential-of-multi","slug":"mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","arxiv_id":"2403.18814","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":5,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dvlab-research/minigemini"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vulnerability-detection-with-code-language","slug":"vulnerability-detection-with-code-language","title":"Vulnerability Detection with Code Language Models: How Far Are We?","date":"2024-03-27","arxiv_id":"2403.18624","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dlvuldet/primevul"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/constructions-are-so-difficult-that-even","slug":"constructions-are-so-difficult-that-even","title":"Constructions Are So Difficult That Even Large Language Models Get Them Right for the Wrong Reasons","date":"2024-03-26","arxiv_id":"2403.17760","n_code_links":1,"syntology":null},{"paper":"/paper/ellen-extremely-lightly-supervised-learning","slug":"ellen-extremely-lightly-supervised-learning","title":"ELLEN: Extremely Lightly Supervised Learning For Efficient Named Entity Recognition","date":"2024-03-26","arxiv_id":"2403.17385","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-legal-document-retrieval-a-multi","title":"Enhancing Legal Document Retrieval: A Multi-Phase Approach with Large Language Models","date":"2024-03-26","arxiv_id":"2403.18093","n_code_links":0,"syntology":null},{"paper":"/paper/internlm2-technical-report","slug":"internlm2-technical-report","title":"InternLM2 Technical Report","date":"2024-03-26","arxiv_id":"2403.17297","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"large-language-models-are-state-of-the-art-2","title":"Large Language Models Are State-of-the-Art Evaluator for Grammatical Error Correction","date":"2024-03-26","arxiv_id":"2403.17540","n_code_links":0,"syntology":null}],"record_sha256":"f64b0164668fe568e227edadef90d4fda0c90fc841abcf76235cda5ef56d3633","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}