{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/18","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":18,"pages_in_order":29,"rows_per_page":100,"rows":[1701,1800],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/17","next":"/method/gpt-4/papers/19","papers":[{"paper":"/paper/the-impact-of-demonstrations-on-multilingual","slug":"the-impact-of-demonstrations-on-multilingual","title":"The Impact of Demonstrations on Multilingual In-Context Learning: A Multidimensional Analysis","date":"2024-02-20","arxiv_id":"2402.12976","n_code_links":1,"syntology":null},{"paper":"/paper/tofueval-evaluating-hallucinations-of-llms-on","slug":"tofueval-evaluating-hallucinations-of-llms-on","title":"TofuEval: Evaluating Hallucinations of LLMs on Topic-Focused Dialogue Summarization","date":"2024-02-20","arxiv_id":"2402.13249","n_code_links":1,"syntology":null},{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/artprompt-ascii-art-based-jailbreak-attacks","slug":"artprompt-ascii-art-based-jailbreak-attacks","title":"ArtPrompt: ASCII Art-based Jailbreak Attacks against Aligned LLMs","date":"2024-02-19","arxiv_id":"2402.11753","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":10,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["uw-nsl/ArtPrompt"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"creating-a-fine-grained-entity-type-taxonomy","title":"Creating a Fine Grained Entity Type Taxonomy Using LLMs","date":"2024-02-19","arxiv_id":"2402.12557","n_code_links":0,"syntology":null},{"paper":null,"slug":"deepcode-ai-fix-fixing-security","title":"DeepCode AI Fix: Fixing Security Vulnerabilities with Large Language Models","date":"2024-02-19","arxiv_id":"2402.13291","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-multilingual-fact-checking-at","title":"Surprising Efficacy of Fine-Tuned Transformers for Fact-Checking over Larger Language Models","date":"2024-02-19","arxiv_id":"2402.12147","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-chatgpt-s-smart-contract","title":"Evaluation of ChatGPT's Smart Contract Auditing Capabilities Based on Chain of Thought","date":"2024-02-19","arxiv_id":"2402.12023","n_code_links":0,"syntology":null},{"paper":"/paper/gtbench-uncovering-the-strategic-reasoning","slug":"gtbench-uncovering-the-strategic-reasoning","title":"GTBench: Uncovering the Strategic Reasoning Limitations of LLMs via Game-Theoretic Evaluations","date":"2024-02-19","arxiv_id":"2402.12348","n_code_links":2,"syntology":null},{"paper":null,"slug":"imbue-improving-interpersonal-effectiveness","title":"IMBUE: Improving Interpersonal Effectiveness through Simulation and Just-in-time Feedback with Human-Language Model Interaction","date":"2024-02-19","arxiv_id":"2402.12556","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-open-source-there-yet-a-comparative-study","title":"Is Open-Source There Yet? A Comparative Study on Commercial and Open-Source LLMs in Their Ability to Label Chest X-Ray Reports","date":"2024-02-19","arxiv_id":"2402.12298","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-ranking-less-capable-language-models-are","title":"Enabling Weak LLMs to Judge Response Reliability via Meta Ranking","date":"2024-02-19","arxiv_id":"2402.12146","n_code_links":0,"syntology":null},{"paper":null,"slug":"mrke-the-multi-hop-reasoning-evaluation-of","title":"Cofca: A Step-Wise Counterfactual Multi-hop QA benchmark","date":"2024-02-19","arxiv_id":"2402.11924","n_code_links":0,"syntology":null},{"paper":"/paper/robust-clip-unsupervised-adversarial-fine","slug":"robust-clip-unsupervised-adversarial-fine","title":"Robust CLIP: Unsupervised Adversarial Fine-Tuning of Vision Embeddings for Robust Large Vision-Language Models","date":"2024-02-19","arxiv_id":"2402.12336","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":1,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chs20/robustvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"shallow-synthesis-of-knowledge-in-gpt","title":"Shallow Synthesis of Knowledge in GPT-Generated Texts: A Case Study in Automatic Related Work Composition","date":"2024-02-19","arxiv_id":"2402.12255","n_code_links":0,"syntology":null},{"paper":null,"slug":"spml-a-dsl-for-defending-language-models","title":"SPML: A DSL for Defending Language Models Against Prompt Attacks","date":"2024-02-19","arxiv_id":"2402.11755","n_code_links":0,"syntology":null},{"paper":"/paper/standardize-aligning-language-models-with","slug":"standardize-aligning-language-models-with","title":"Standardize: Aligning Language Models with Expert-Defined Standards for Content Generation","date":"2024-02-19","arxiv_id":"2402.12593","n_code_links":1,"syntology":null},{"paper":null,"slug":"your-large-language-model-is-secretly-a","title":"Your Large Language Model is Secretly a Fairness Proponent and You Should Prompt it Like One","date":"2024-02-19","arxiv_id":"2402.12150","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-deception-detection-go-deeper-dataset","title":"Can Deception Detection Go Deeper? Dataset, Evaluation, and Benchmark for Deception Reasoning","date":"2024-02-18","arxiv_id":"2402.11432","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-news-narratives-a-critical-analysis","title":"Decoding News Narratives: A Critical Analysis of Large Language Models in Framing Detection","date":"2024-02-18","arxiv_id":"2402.11621","n_code_links":0,"syntology":null},{"paper":null,"slug":"dictllm-harnessing-key-value-data-structures","title":"DictLLM: Harnessing Key-Value Data Structures with Large Language Models for Enhanced Medical Diagnostics","date":"2024-02-18","arxiv_id":"2402.11481","n_code_links":0,"syntology":null},{"paper":"/paper/eventrl-enhancing-event-extraction-with","slug":"eventrl-enhancing-event-extraction-with","title":"EventRL: Enhancing Event Extraction with Outcome Supervision for Large Language Models","date":"2024-02-18","arxiv_id":"2402.11430","n_code_links":1,"syntology":null},{"paper":"/paper/factpico-factuality-evaluation-for-plain","slug":"factpico-factuality-evaluation-for-plain","title":"FactPICO: Factuality Evaluation for Plain Language Summarization of Medical Evidence","date":"2024-02-18","arxiv_id":"2402.11456","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lilywchen/factpico"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kmmlu-measuring-massive-multitask-language","title":"KMMLU: Measuring Massive Multitask Language Understanding in Korean","date":"2024-02-18","arxiv_id":"2402.11548","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-failure-integrating-negative","slug":"learning-from-failure-integrating-negative","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","date":"2024-02-18","arxiv_id":"2402.11651","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["reason-wang/nat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/longagent-scaling-language-models-to-128k","slug":"longagent-scaling-language-models-to-128k","title":"LongAgent: Scaling Language Models to 128k Context through Multi-Agent Collaboration","date":"2024-02-18","arxiv_id":"2402.11550","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-dimensional-evaluation-of-empathetic","title":"Multi-dimensional Evaluation of Empathetic Dialog Responses","date":"2024-02-18","arxiv_id":"2402.11409","n_code_links":0,"syntology":null},{"paper":"/paper/multi-task-inference-can-large-language","slug":"multi-task-inference-can-large-language","title":"Multi-Task Inference: Can Large Language Models Follow Multiple Instructions at Once?","date":"2024-02-18","arxiv_id":"2402.11597","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["guijinson/mti-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ploutos-towards-interpretable-stock-movement","title":"Ploutos: Towards interpretable stock movement prediction with financial large language model","date":"2024-02-18","arxiv_id":"2403.00782","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-flan-scaling-human-labeled-tasks-in","title":"Vision-Flan: Scaling Human-Labeled Tasks in Visual Instruction Tuning","date":"2024-02-18","arxiv_id":"2402.11690","n_code_links":0,"syntology":null},{"paper":"/paper/boosting-of-thoughts-trial-and-error-problem","slug":"boosting-of-thoughts-trial-and-error-problem","title":"Boosting of Thoughts: Trial-and-Error Problem Solving with Large Language Models","date":"2024-02-17","arxiv_id":"2402.11140","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-chatgpt-for-next-generation","title":"Exploring ChatGPT for Next-generation Information Retrieval: Opportunities and Challenges","date":"2024-02-17","arxiv_id":"2402.11203","n_code_links":0,"syntology":null},{"paper":null,"slug":"gendec-a-robust-generative-question","title":"GenDec: A robust generative Question-decomposition method for Multi-hop reasoning","date":"2024-02-17","arxiv_id":"2402.11166","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoning-before-comparison-llm-enhanced","title":"Reasoning before Comparison: LLM-Enhanced Semantic Similarity Metrics for Domain Specialized Text Analysis","date":"2024-02-17","arxiv_id":"2402.11398","n_code_links":0,"syntology":null},{"paper":"/paper/zerog-investigating-cross-dataset-zero-shot","slug":"zerog-investigating-cross-dataset-zero-shot","title":"ZeroG: Investigating Cross-dataset Zero-shot Transferability in Graphs","date":"2024-02-17","arxiv_id":"2402.11235","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nineabyss/zerog"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-the-reasoning-abilities-of-chatgpt","title":"Assessing the Reasoning Abilities of ChatGPT in the Context of Claim Verification","date":"2024-02-16","arxiv_id":"2402.10735","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-speak-for-diverse-people-tuning-llms","slug":"can-llms-speak-for-diverse-people-tuning-llms","title":"Can LLMs Speak For Diverse People? Tuning LLMs via Debate to Generate Controllable Controversial Statements","date":"2024-02-16","arxiv_id":"2402.10614","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tianyi-lab/debatune"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-separators-improve-chain-of-thought","title":"Can Separators Improve Chain-of-Thought Prompting?","date":"2024-02-16","arxiv_id":"2402.10645","n_code_links":0,"syntology":null},{"paper":null,"slug":"emoji-driven-crypto-assets-market-reactions","title":"Emoji Driven Crypto Assets Market Reactions","date":"2024-02-16","arxiv_id":"2402.10481","n_code_links":0,"syntology":null},{"paper":null,"slug":"fintral-a-family-of-gpt-4-level-multimodal","title":"FinTral: A Family of GPT-4 Level Multimodal Financial Large Language Models","date":"2024-02-16","arxiv_id":"2402.10986","n_code_links":0,"syntology":null},{"paper":"/paper/german-text-simplification-finetuning-large","slug":"german-text-simplification-finetuning-large","title":"German Text Simplification: Finetuning Large Language Models with Semi-Synthetic Data","date":"2024-02-16","arxiv_id":"2402.10675","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-reliable-are-automatic-evaluation-methods","title":"How Reliable Are Automatic Evaluation Methods for Instruction-Tuned LLMs?","date":"2024-02-16","arxiv_id":"2402.10770","n_code_links":0,"syntology":null},{"paper":"/paper/in-search-of-needles-in-a-10m-haystack","slug":"in-search-of-needles-in-a-10m-haystack","title":"In Search of Needles in a 11M Haystack: Recurrent Memory Finds What LLMs Miss","date":"2024-02-16","arxiv_id":"2402.10790","n_code_links":2,"syntology":null},{"paper":"/paper/jailbreaking-proprietary-large-language","slug":"jailbreaking-proprietary-large-language","title":"When \"Competency\" in Reasoning Opens the Door to Vulnerability: Jailbreaking LLMs via Novel Complex Ciphers","date":"2024-02-16","arxiv_id":"2402.10601","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["divijh/jailbreak_cryptography"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-as-zero-shot-dialogue","slug":"large-language-models-as-zero-shot-dialogue","title":"Large Language Models as Zero-shot Dialogue State Tracker through Function Calling","date":"2024-02-16","arxiv_id":"2402.10466","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["facebookresearch/fnctod"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"large-language-models-fall-short","title":"Large Language Models Fall Short: Understanding Complex Relationships in Detective Narratives","date":"2024-02-16","arxiv_id":"2402.11051","n_code_links":0,"syntology":null},{"paper":"/paper/linkner-linking-local-named-entity","slug":"linkner-linking-local-named-entity","title":"LinkNER: Linking Local Named Entity Recognition Models to Large Language Models using Uncertainty","date":"2024-02-16","arxiv_id":"2402.10573","n_code_links":1,"syntology":null},{"paper":"/paper/toolsword-unveiling-safety-issues-of-large","slug":"toolsword-unveiling-safety-issues-of-large","title":"ToolSword: Unveiling Safety Issues of Large Language Models in Tool Learning Across Three Stages","date":"2024-02-16","arxiv_id":"2402.10753","n_code_links":1,"syntology":null},{"paper":"/paper/a-strongreject-for-empty-jailbreaks","slug":"a-strongreject-for-empty-jailbreaks","title":"A StrongREJECT for Empty Jailbreaks","date":"2024-02-15","arxiv_id":"2402.10260","n_code_links":2,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["alexandrasouly/strongreject"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":null,"slug":"an-analysis-of-langauge-frequency-and-error","title":"An Analysis of Language Frequency and Error Correction for Esperanto","date":"2024-02-15","arxiv_id":"2402.09696","n_code_links":0,"syntology":null},{"paper":"/paper/data-engineering-for-scaling-language-models","slug":"data-engineering-for-scaling-language-models","title":"Data Engineering for Scaling Language Models to 128K Context","date":"2024-02-15","arxiv_id":"2402.10171","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/long-context-data-engineering"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-large-language-model-llm","title":"Fine-tuning Large Language Model (LLM) Artificial Intelligence Chatbots in Ophthalmology and LLM-based evaluation using GPT-4","date":"2024-02-15","arxiv_id":"2402.10083","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-s-assessment-of-its-performance-in-a","title":"GPT-4's assessment of its performance in a USMLE-based case study","date":"2024-02-15","arxiv_id":"2402.09654","n_code_links":0,"syntology":null},{"paper":"/paper/openmathinstruct-1-a-1-8-million-math","slug":"openmathinstruct-1-a-1-8-million-math","title":"OpenMathInstruct-1: A 1.8 Million Math Instruction Tuning Dataset","date":"2024-02-15","arxiv_id":"2402.10176","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-based-bias-calibration-for-better-zero","title":"Prompt-Based Bias Calibration for Better Zero/Few-Shot Learning of Language Models","date":"2024-02-15","arxiv_id":"2402.10353","n_code_links":0,"syntology":null},{"paper":"/paper/unlocking-structure-measuring-introducing-pdd","slug":"unlocking-structure-measuring-introducing-pdd","title":"Unlocking Structure Measuring: Introducing PDD, an Automatic Metric for Positional Discourse Coherence","date":"2024-02-15","arxiv_id":"2402.10175","n_code_links":1,"syntology":null},{"paper":null,"slug":"x-lifecycle-learning-for-cloud-incident","title":"X-lifecycle Learning for Cloud Incident Management using LLMs","date":"2024-02-15","arxiv_id":"2404.03662","n_code_links":0,"syntology":null},{"paper":"/paper/api-pack-a-massive-multilingual-dataset-for","slug":"api-pack-a-massive-multilingual-dataset-for","title":"API Pack: A Massive Multi-Programming Language Dataset for API Call Generation","date":"2024-02-14","arxiv_id":"2402.09615","n_code_links":1,"syntology":null},{"paper":"/paper/aqa-bench-an-interactive-benchmark-for","slug":"aqa-bench-an-interactive-benchmark-for","title":"AQA-Bench: An Interactive Benchmark for Evaluating LLMs' Sequential Reasoning Ability","date":"2024-02-14","arxiv_id":"2402.09404","n_code_links":1,"syntology":null},{"paper":null,"slug":"l3go-language-agents-with-chain-of-3d","title":"L3GO: Language Agents with Chain-of-3D-Thoughts for Generating Unconventional Objects","date":"2024-02-14","arxiv_id":"2402.09052","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-enhanced-1","title":"Leveraging Large Language Models for Enhanced NLP Task Performance through Knowledge Distillation and Optimized Training Strategies","date":"2024-02-14","arxiv_id":"2402.09282","n_code_links":0,"syntology":null},{"paper":"/paper/llasmol-advancing-large-language-models-for","slug":"llasmol-advancing-large-language-models-for","title":"LlaSMol: Advancing Large Language Models for Chemistry with a Large-Scale, Comprehensive, High-Quality Instruction Tuning Dataset","date":"2024-02-14","arxiv_id":"2402.09391","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["osu-nlp-group/llm4chem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-the-authoring-of-autotutors-with","slug":"scaling-the-authoring-of-autotutors-with","title":"AutoTutor meets Large Language Models: A Language Model Tutor with Rich Pedagogy and Guardrails","date":"2024-02-14","arxiv_id":"2402.09216","n_code_links":1,"syntology":{"ran":4,"of":10,"n_ran_checked":4,"n_instrument":0,"unverified":6,"pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["eth-lre/mwptutor"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/bbox-adapter-lightweight-adapting-for-black","slug":"bbox-adapter-lightweight-adapting-for-black","title":"BBox-Adapter: Lightweight Adapting for Black-Box Large Language Models","date":"2024-02-13","arxiv_id":"2402.08219","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haotiansun14/bbox-adapter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"combining-insights-from-multiple-large","title":"Combining Insights From Multiple Large Language Models Improves Diagnostic Accuracy","date":"2024-02-13","arxiv_id":"2402.08806","n_code_links":0,"syntology":null},{"paper":"/paper/ecellm-generalizing-large-language-models-for","slug":"ecellm-generalizing-large-language-models-for","title":"eCeLLM: Generalizing Large Language Models for E-commerce from Large-scale, High-quality Instruction Data","date":"2024-02-13","arxiv_id":"2402.08831","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ninglab/ecellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/instructgraph-boosting-large-language-models","slug":"instructgraph-boosting-large-language-models","title":"InstructGraph: Boosting Large Language Models via Graph-centric Instruction Tuning and Preference Alignment","date":"2024-02-13","arxiv_id":"2402.08785","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-for-the-automated","slug":"large-language-models-for-the-automated","title":"Large Language Models for the Automated Analysis of Optimization Algorithms","date":"2024-02-13","arxiv_id":"2402.08472","n_code_links":1,"syntology":null},{"paper":"/paper/llaga-large-language-and-graph-assistant","slug":"llaga-large-language-and-graph-assistant","title":"LLaGA: Large Language and Graph Assistant","date":"2024-02-13","arxiv_id":"2402.08170","n_code_links":2,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chenrunjin/llaga","vita-group/llaga"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompt-optimization-in-multi-step-tasks","slug":"prompt-optimization-in-multi-step-tasks","title":"PRompt Optimization in Multi-Step Tasks (PROMST): Integrating Human Feedback and Heuristic-based Sampling","date":"2024-02-13","arxiv_id":"2402.08702","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yongchao98/promst"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-last-jitai-the-unreasonable-effectiveness","title":"The Last JITAI? Exploring Large Language Models for Issuing Just-in-Time Adaptive Interventions: Fostering Physical Activity in a Conceptual Cardiac Rehabilitation Setting","date":"2024-02-13","arxiv_id":"2402.08658","n_code_links":0,"syntology":null},{"paper":"/paper/addressing-cognitive-bias-in-medical-language","slug":"addressing-cognitive-bias-in-medical-language","title":"Addressing cognitive bias in medical language models","date":"2024-02-12","arxiv_id":"2402.08113","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["carlwharris/cog-bias-med-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","n_code_links":1,"syntology":null},{"paper":"/paper/dolares-or-dollars-unraveling-the-bilingual","slug":"dolares-or-dollars-unraveling-the-bilingual","title":"Dólares or Dollars? Unraveling the Bilingual Prowess of Financial LLMs Between Spanish and English","date":"2024-02-12","arxiv_id":"2402.07405","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-multi-criteria-decision-analysis","title":"Enhancing Multi-Criteria Decision Analysis with AI: Integrating Analytic Hierarchy Process and GPT-4 for Automated Decision Support","date":"2024-02-12","arxiv_id":"2402.07404","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-programming-error-messages-in-real","title":"Enhancing Programming Error Messages in Real Time with Generative AI","date":"2024-02-12","arxiv_id":"2402.08072","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-ad-referendum-how-good","title":"Large Language Models \"Ad Referendum\": How Good Are They at Machine Translation in the Legal Domain?","date":"2024-02-12","arxiv_id":"2402.07681","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-few-shot-generators","slug":"large-language-models-are-few-shot-generators","title":"Large Language Models are Few-shot Generators: Proposing Hybrid Prompt Algorithm To Generate Webshell Escape Samples","date":"2024-02-12","arxiv_id":"2402.07408","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-ai-to-advance-science-and","title":"Leveraging AI to Advance Science and Computing Education across Africa: Challenges, Progress and Opportunities","date":"2024-02-12","arxiv_id":"2402.07397","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-self-verification-limitations-of-large","title":"On the Self-Verification Limitations of Large Language Models on Reasoning and Planning Tasks","date":"2024-02-12","arxiv_id":"2402.08115","n_code_links":0,"syntology":null},{"paper":null,"slug":"secret-collusion-among-generative-ai-agents","title":"Secret Collusion among Generative AI Agents: Multi-Agent Deception via Steganography","date":"2024-02-12","arxiv_id":"2402.07510","n_code_links":0,"syntology":null},{"paper":null,"slug":"suppressing-pink-elephants-with-direct","title":"Suppressing Pink Elephants with Direct Principle Feedback","date":"2024-02-12","arxiv_id":"2402.07896","n_code_links":0,"syntology":null},{"paper":"/paper/vcr-video-representation-for-contextual","slug":"vcr-video-representation-for-contextual","title":"VCR: Video representation for Contextual Retrieval","date":"2024-02-12","arxiv_id":"2402.07466","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-do-large-language-models-navigate","title":"How do Large Language Models Navigate Conflicts between Honesty and Helpfulness?","date":"2024-02-11","arxiv_id":"2402.07282","n_code_links":0,"syntology":null},{"paper":null,"slug":"natural-language-reinforcement-learning","title":"Natural Language Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07157","n_code_links":0,"syntology":null},{"paper":"/paper/chemllm-a-chemical-large-language-model","slug":"chemllm-a-chemical-large-language-model","title":"ChemLLM: A Chemical Large Language Model","date":"2024-02-10","arxiv_id":"2402.06852","n_code_links":1,"syntology":null},{"paper":"/paper/gemini-goes-to-med-school-exploring-the","slug":"gemini-goes-to-med-school-exploring-the","title":"Gemini Goes to Med School: Exploring the Capabilities of Multimodal Large Language Models on Medical Challenge Problems & Hallucinations","date":"2024-02-10","arxiv_id":"2402.07023","n_code_links":1,"syntology":null},{"paper":"/paper/openfedllm-training-large-language-models-on","slug":"openfedllm-training-large-language-models-on","title":"OpenFedLLM: Training Large Language Models on Decentralized Private Data via Federated Learning","date":"2024-02-10","arxiv_id":"2402.06954","n_code_links":3,"syntology":null},{"paper":"/paper/urbankgent-a-unified-large-language-model","slug":"urbankgent-a-unified-large-language-model","title":"UrbanKGent: A Unified Large Language Model Agent Framework for Urban Knowledge Graph Construction","date":"2024-02-10","arxiv_id":"2402.06861","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["usail-hkust/urbankgent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/bryndza-at-climateactivism-2024-stance-target","slug":"bryndza-at-climateactivism-2024-stance-target","title":"Bryndza at ClimateActivism 2024: Stance, Target and Hate Event Detection via Retrieval-Augmented GPT-4 and LLaMA","date":"2024-02-09","arxiv_id":"2402.06549","n_code_links":2,"syntology":null},{"paper":"/paper/culturellm-incorporating-cultural-differences","slug":"culturellm-incorporating-cultural-differences","title":"CultureLLM: Incorporating Cultural Differences into Large Language Models","date":"2024-02-09","arxiv_id":"2402.10946","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["scarelette/culturellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"llava-docent-instruction-tuning-with","title":"LLaVA-Docent: Instruction Tuning with Multimodal Large Language Model to Support Art Appreciation Education","date":"2024-02-09","arxiv_id":"2402.06264","n_code_links":0,"syntology":null},{"paper":"/paper/rarebench-can-llms-serve-as-rare-diseases","slug":"rarebench-can-llms-serve-as-rare-diseases","title":"RareBench: Can LLMs Serve as Rare Diseases Specialists?","date":"2024-02-09","arxiv_id":"2402.06341","n_code_links":1,"syntology":null},{"paper":"/paper/resumeflow-an-llm-facilitated-pipeline-for","slug":"resumeflow-an-llm-facilitated-pipeline-for","title":"ResumeFlow: An LLM-facilitated Pipeline for Personalized Resume Generation and Refinement","date":"2024-02-09","arxiv_id":"2402.06221","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Ztrimus/job-llm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-prompt-response-to-the-demand-for-automatic","slug":"a-prompt-response-to-the-demand-for-automatic","title":"A Prompt Response to the Demand for Automatic Gender-Neutral Translation","date":"2024-02-08","arxiv_id":"2402.06041","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-4-generated-narratives-of-life-events","title":"GPT-4 Generated Narratives of Life Events using a Structured Narrative Prompt: A Validation Study","date":"2024-02-08","arxiv_id":"2402.05435","n_code_links":0,"syntology":null},{"paper":"/paper/how-well-can-llms-negotiate-negotiationarena","slug":"how-well-can-llms-negotiate-negotiationarena","title":"How Well Can LLMs Negotiate? NegotiationArena Platform and Analysis","date":"2024-02-08","arxiv_id":"2402.05863","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vinid/negotiationarena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-psycholinguistic","title":"Large Language Models for Psycholinguistic Plausibility Pretesting","date":"2024-02-08","arxiv_id":"2402.05455","n_code_links":0,"syntology":null},{"paper":"/paper/limits-of-transformer-language-models-on","slug":"limits-of-transformer-language-models-on","title":"Limits of Transformer Language Models on Learning to Compose Algorithms","date":"2024-02-08","arxiv_id":"2402.05785","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":3,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ibm/limitations-lm-algorithmic-compositional-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}}],"record_sha256":"e85a0d076a67f97f6dccb6d547cf4d2397aba248ad78e61ea21a1fa2bf648c8d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}