{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/14","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":14,"pages_in_order":20,"rows_per_page":100,"rows":[1301,1400],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/13","next":"/method/gpt-3/papers/15","papers":[{"paper":"/paper/towards-explainable-conversational","slug":"towards-explainable-conversational","title":"Towards Explainable Conversational Recommender Systems","date":"2023-05-27","arxiv_id":"2305.18363","n_code_links":1,"syntology":null},{"paper":"/paper/what-can-large-language-models-do-in","slug":"what-can-large-language-models-do-in","title":"What can Large Language Models do in chemistry? A comprehensive benchmark on eight tasks","date":"2023-05-27","arxiv_id":"2305.18365","n_code_links":1,"syntology":null},{"paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","slug":"chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","arxiv_id":"2305.17306","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/chain-of-thought-hub"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatgpt-a-study-on-its-utility-for-ubiquitous","title":"ChatGPT: A Study on its Utility for Ubiquitous Software Engineering Tasks","date":"2023-05-26","arxiv_id":"2305.16837","n_code_links":0,"syntology":null},{"paper":"/paper/counterfactual-reasoning-testing-language","slug":"counterfactual-reasoning-testing-language","title":"Counterfactual reasoning: Testing language models' understanding of hypothetical scenarios","date":"2023-05-26","arxiv_id":"2305.16572","n_code_links":1,"syntology":null},{"paper":null,"slug":"distinguishing-human-generated-text-from","title":"Distinguishing Human Generated Text From ChatGPT Generated Text Using Machine Learning","date":"2023-05-26","arxiv_id":"2306.01761","n_code_links":0,"syntology":null},{"paper":"/paper/do-gpts-produce-less-literal-translations","slug":"do-gpts-produce-less-literal-translations","title":"Do GPTs Produce Less Literal Translations?","date":"2023-05-26","arxiv_id":"2305.16806","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-question-generation-needs-more","title":"Evaluation of Question Generation Needs More References","date":"2023-05-26","arxiv_id":"2305.16626","n_code_links":0,"syntology":null},{"paper":null,"slug":"impossible-distillation-from-low-quality","title":"Impossible Distillation: from Low-Quality Model to High-Quality Dataset & Model for Summarization and Paraphrasing","date":"2023-05-26","arxiv_id":"2305.16635","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-accuracy-of-gpt-3-4-results-on","title":"Improving accuracy of GPT-3/4 results on biomedical data using a retrieval-augmented language model","date":"2023-05-26","arxiv_id":"2305.17116","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-as-tool-makers","slug":"large-language-models-as-tool-makers","title":"Large Language Models as Tool Makers","date":"2023-05-26","arxiv_id":"2305.17126","n_code_links":1,"syntology":null},{"paper":null,"slug":"playing-repeated-games-with-large-language","title":"Playing repeated games with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16867","n_code_links":0,"syntology":null},{"paper":"/paper/linguistic-properties-of-truthful-response","slug":"linguistic-properties-of-truthful-response","title":"Linguistic Properties of Truthful Response","date":"2023-05-25","arxiv_id":"2305.15875","n_code_links":1,"syntology":null},{"paper":null,"slug":"not-wacky-vs-definitely-wacky-a-study-of","title":"Not wacky vs. definitely wacky: A study of scalar adverbs in pretrained language models","date":"2023-05-25","arxiv_id":"2305.16426","n_code_links":0,"syntology":null},{"paper":"/paper/a-causal-view-of-entity-bias-in-large","slug":"a-causal-view-of-entity-bias-in-large","title":"A Causal View of Entity Bias in (Large) Language Models","date":"2023-05-24","arxiv_id":"2305.14695","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luka-group/causal-view-of-entity-bias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmarking-arabic-ai-with-large-language","title":"LAraBench: Benchmarking Arabic AI with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14982","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-questions-training-with-latent","title":"Chain-of-Questions Training with Latent Answers for Robust Multistep Question Answering","date":"2023-05-24","arxiv_id":"2305.14901","n_code_links":0,"syntology":null},{"paper":"/paper/chatagri-exploring-potentials-of-chatgpt-on","slug":"chatagri-exploring-potentials-of-chatgpt-on","title":"ChatAgri: Exploring Potentials of ChatGPT on Cross-linguistic Agricultural Text Classification","date":"2023-05-24","arxiv_id":"2305.15024","n_code_links":1,"syntology":null},{"paper":null,"slug":"don-t-take-this-out-of-context-on-the-need","title":"Don't Take This Out of Context! On the Need for Contextual Models and Evaluations for Stylistic Rewriting","date":"2023-05-24","arxiv_id":"2305.14755","n_code_links":0,"syntology":null},{"paper":"/paper/expertprompting-instructing-large-language","slug":"expertprompting-instructing-large-language","title":"ExpertPrompting: Instructing Large Language Models to be Distinguished Experts","date":"2023-05-24","arxiv_id":"2305.14688","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ofa-sys/expertllama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/harnessing-the-power-of-large-language-models","slug":"harnessing-the-power-of-large-language-models","title":"Harnessing the Power of Large Language Models for Natural Language to First-Order Logic Translation","date":"2023-05-24","arxiv_id":"2305.15541","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gblackout/logicllama"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-centered-metrics-for-dialog-system","title":"Psychological Metrics for Dialog System Evaluation","date":"2023-05-24","arxiv_id":"2305.14757","n_code_links":0,"syntology":null},{"paper":"/paper/i-spy-a-metaphor-large-language-models-and","slug":"i-spy-a-metaphor-large-language-models-and","title":"I Spy a Metaphor: Large Language Models and Diffusion Models Co-Create Visual Metaphors","date":"2023-05-24","arxiv_id":"2305.14724","n_code_links":1,"syntology":null},{"paper":"/paper/inference-time-policy-adapters-ipa-tailoring","slug":"inference-time-policy-adapters-ipa-tailoring","title":"Inference-Time Policy Adapters (IPA): Tailoring Extreme-Scale LMs without Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15065","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":5,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gximinglu/ipa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-llms-for-kpis-retrieval-from","title":"Enabling and Analyzing How to Efficiently Extract Information from Hybrid Long Documents with LLMs","date":"2023-05-24","arxiv_id":"2305.16344","n_code_links":0,"syntology":null},{"paper":null,"slug":"mastering-the-abcds-of-complex-questions","title":"Mastering the ABCDs of Complex Questions: Answer-Based Claim Decomposition for Fine-grained Self-Evaluation","date":"2023-05-24","arxiv_id":"2305.14750","n_code_links":0,"syntology":null},{"paper":"/paper/peek-across-improving-multi-document-modeling","slug":"peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","arxiv_id":"2305.15387","n_code_links":1,"syntology":{"ran":12,"of":18,"n_ran_checked":11,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["aviclu/peekacross"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-checker-plug-and-play-modules-for-fact","slug":"self-checker-plug-and-play-modules-for-fact","title":"Self-Checker: Plug-and-Play Modules for Fact-Checking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14623","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Miaoranmmm/SelfChecker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/testing-causal-models-of-word-meaning-in-gpt","slug":"testing-causal-models-of-word-meaning-in-gpt","title":"Testing Causal Models of Word Meaning in GPT-3 and -4","date":"2023-05-24","arxiv_id":"2305.14630","n_code_links":1,"syntology":null},{"paper":"/paper/tomchallenges-a-principle-guided-dataset-and","slug":"tomchallenges-a-principle-guided-dataset-and","title":"ToMChallenges: A Principle-Guided Dataset and Diverse Evaluation Tasks for Exploring Theory of Mind","date":"2023-05-24","arxiv_id":"2305.15068","n_code_links":1,"syntology":null},{"paper":"/paper/complementing-gpt-3-with-few-shot-sequence-to","slug":"complementing-gpt-3-with-few-shot-sequence-to","title":"Fine-tuned LLMs Know More, Hallucinate Less with Few-Shot Sequence-to-Sequence Semantic Parsing over Wikidata","date":"2023-05-23","arxiv_id":"2305.14202","n_code_links":1,"syntology":null},{"paper":null,"slug":"dancing-between-success-and-failure-edit","title":"Dancing Between Success and Failure: Edit-level Simplification Evaluation using SALSA","date":"2023-05-23","arxiv_id":"2305.14458","n_code_links":0,"syntology":null},{"paper":"/paper/dynosaur-a-dynamic-growth-paradigm-for","slug":"dynosaur-a-dynamic-growth-paradigm-for","title":"Dynosaur: A Dynamic Growth Paradigm for Instruction-Tuning Data Curation","date":"2023-05-23","arxiv_id":"2305.14327","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":10,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wadeyin9712/dynosaur"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"efficient-open-domain-multi-hop-question","title":"Few-Shot Data Synthesis for Open Domain Multi-Hop Question Answering","date":"2023-05-23","arxiv_id":"2305.13691","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-black-box-few-shot-text","title":"Enhancing Black-Box Few-Shot Text Classification with Prompt-Based Data Augmentation","date":"2023-05-23","arxiv_id":"2305.13785","n_code_links":0,"syntology":null},{"paper":"/paper/how-old-is-gpt-the-humbel-framework-for","slug":"how-old-is-gpt-the-humbel-framework-for","title":"HumBEL: A Human-in-the-Loop Approach for Evaluating Demographic Factors of Language Models in Human-Machine Conversations","date":"2023-05-23","arxiv_id":"2305.14195","n_code_links":1,"syntology":null},{"paper":null,"slug":"ifqa-a-dataset-for-open-domain-question","title":"IfQA: A Dataset for Open-domain Question Answering under Counterfactual Presuppositions","date":"2023-05-23","arxiv_id":"2305.14010","n_code_links":0,"syntology":null},{"paper":"/paper/images-in-language-space-exploring-the","slug":"images-in-language-space-exploring-the","title":"Images in Language Space: Exploring the Suitability of Large Language Models for Vision & Language Tasks","date":"2023-05-23","arxiv_id":"2305.13782","n_code_links":1,"syntology":null},{"paper":"/paper/instructscore-towards-explainable-text","slug":"instructscore-towards-explainable-text","title":"INSTRUCTSCORE: Explainable Text Generation Evaluation with Finegrained Feedback","date":"2023-05-23","arxiv_id":"2305.14282","n_code_links":2,"syntology":null},{"paper":"/paper/let-s-think-frame-by-frame-evaluating-video","slug":"let-s-think-frame-by-frame-evaluating-video","title":"Let's Think Frame by Frame with VIP: A Video Infilling and Prediction Dataset for Evaluating Video Chain-of-Thought","date":"2023-05-23","arxiv_id":"2305.13903","n_code_links":1,"syntology":null},{"paper":"/paper/mathdial-a-dialogue-tutoring-dataset-with","slug":"mathdial-a-dialogue-tutoring-dataset-with","title":"MathDial: A Dialogue Tutoring Dataset with Rich Pedagogical Properties Grounded in Math Reasoning Problems","date":"2023-05-23","arxiv_id":"2305.14536","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["eth-nlped/mathdial"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nail-lexical-retrieval-indices-with-efficient","title":"NAIL: Lexical Retrieval Indices with Efficient Non-Autoregressive Decoders","date":"2023-05-23","arxiv_id":"2305.14499","n_code_links":0,"syntology":null},{"paper":"/paper/sources-of-hallucination-by-large-language","slug":"sources-of-hallucination-by-large-language","title":"Sources of Hallucination by Large Language Models on Inference Tasks","date":"2023-05-23","arxiv_id":"2305.14552","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-failures-of-self-consistency-in-the-multi","title":"Two Failures of Self-Consistency in the Multi-Step Reasoning of LLMs","date":"2023-05-23","arxiv_id":"2305.14279","n_code_links":0,"syntology":null},{"paper":"/paper/wikichat-a-few-shot-llm-based-chatbot","slug":"wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","arxiv_id":"2305.14292","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["stanford-oval/wikichat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2305-14386","title":"Let GPT be a Math Tutor: Teaching Math Word Problem Solvers with Customized Exercise Generation","date":"2023-05-22","arxiv_id":"2305.14386","n_code_links":0,"syntology":null},{"paper":"/paper/a-study-of-generative-large-language-model","slug":"a-study-of-generative-large-language-model","title":"A Study of Generative Large Language Model for Medical Research and Healthcare","date":"2023-05-22","arxiv_id":"2305.13523","n_code_links":1,"syntology":null},{"paper":null,"slug":"cognitive-network-science-reveals-bias-in-gpt","title":"Cognitive network science reveals bias in GPT-3, ChatGPT, and GPT-4 mirroring math anxiety in high-school students","date":"2023-05-22","arxiv_id":"2305.18320","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-and-enhancing-structural","slug":"evaluating-and-enhancing-structural","title":"Table Meets LLM: Can Large Language Models Understand Structured Table Data? A Benchmark and Empirical Study","date":"2023-05-22","arxiv_id":"2305.13062","n_code_links":1,"syntology":null},{"paper":null,"slug":"inheritsumm-a-general-versatile-and-compact","title":"InheritSumm: A General, Versatile and Compact Summarizer by Distilling from GPT","date":"2023-05-22","arxiv_id":"2305.13083","n_code_links":0,"syntology":null},{"paper":"/paper/mailex-email-event-and-argument-extraction","slug":"mailex-email-event-and-argument-extraction","title":"MAILEX: Email Event and Argument Extraction","date":"2023-05-22","arxiv_id":"2305.13469","n_code_links":1,"syntology":null},{"paper":"/paper/measuring-inductive-biases-of-in-context","slug":"measuring-inductive-biases-of-in-context","title":"Measuring Inductive Biases of In-Context Learning with Underspecified Demonstrations","date":"2023-05-22","arxiv_id":"2305.13299","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-dialogue-systems-with-agency-in-human","title":"Investigating Agency of LLMs in Human-AI Collaboration Tasks","date":"2023-05-22","arxiv_id":"2305.12815","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-symbolic-framework-for-systematic","title":"A Symbolic Framework for Evaluating Mathematical Reasoning and Generalisation with Transformers","date":"2023-05-21","arxiv_id":"2305.12563","n_code_links":0,"syntology":null},{"paper":"/paper/biasasker-measuring-the-bias-in","slug":"biasasker-measuring-the-bias-in","title":"BiasAsker: Measuring the Bias in Conversational AI System","date":"2023-05-21","arxiv_id":"2305.12434","n_code_links":1,"syntology":null},{"paper":"/paper/contrastive-learning-with-logic-driven-data","slug":"contrastive-learning-with-logic-driven-data","title":"Abstract Meaning Representation-Based Logic-Driven Data Augmentation for Logical Reasoning","date":"2023-05-21","arxiv_id":"2305.12599","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-3-5-vs-gpt-4-evaluating-chatgpt-s","title":"GPT-3.5, GPT-4, or BARD? Evaluating LLMs Reasoning Ability in Zero-Shot Setting and Performance Boosting Through Prompts","date":"2023-05-21","arxiv_id":"2305.12477","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-nlp-models-correctly-reason-over-contexts","title":"Can NLP Models Correctly Reason Over Contexts that Break the Common Assumptions?","date":"2023-05-20","arxiv_id":"2305.12096","n_code_links":0,"syntology":null},{"paper":"/paper/logicot-logical-chain-of-thought-instruction","slug":"logicot-logical-chain-of-thought-instruction","title":"LogiCoT: Logical Chain-of-Thought Instruction-Tuning","date":"2023-05-20","arxiv_id":"2305.12147","n_code_links":1,"syntology":null},{"paper":null,"slug":"practical-pcg-through-large-language-models","title":"Practical PCG Through Large Language Models","date":"2023-05-20","arxiv_id":"2305.18243","n_code_links":0,"syntology":null},{"paper":null,"slug":"autotrial-prompting-language-models-for","title":"AutoTrial: Prompting Language Models for Clinical Trial Design","date":"2023-05-19","arxiv_id":"2305.11366","n_code_links":0,"syntology":null},{"paper":"/paper/clinical-camel-an-open-source-expert-level","slug":"clinical-camel-an-open-source-expert-level","title":"Clinical Camel: An Open Expert-Level Medical Language Model with Dialogue-Based Knowledge Encoding","date":"2023-05-19","arxiv_id":"2305.12031","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bowang-lab/clinical-camel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-the-upper-limits-of-text-based","title":"Exploring the Upper Limits of Text-Based Collaborative Filtering Using Large Language Models: Discoveries and Insights","date":"2023-05-19","arxiv_id":"2305.11700","n_code_links":0,"syntology":null},{"paper":"/paper/seegull-a-stereotype-benchmark-with-broad-geo","slug":"seegull-a-stereotype-benchmark-with-broad-geo","title":"SeeGULL: A Stereotype Benchmark with Broad Geo-Cultural Coverage Leveraging Generative Models","date":"2023-05-19","arxiv_id":"2305.11840","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-agreement-a-framework-for-fine-tuning","title":"Self-Agreement: A Framework for Fine-tuning Language Models to Find Agreement among Diverse Opinions","date":"2023-05-19","arxiv_id":"2305.11460","n_code_links":0,"syntology":null},{"paper":null,"slug":"aiwriting-relations-between-image-generation","title":"AIwriting: Relations Between Image Generation and Digital Writing","date":"2023-05-18","arxiv_id":"2305.10834","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-methods-for-extracting","title":"Deep Learning Methods for Extracting Metaphorical Names of Flowers and Plants","date":"2023-05-18","arxiv_id":"2305.10833","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-multiple-intent-conditioned-slot","title":"Generalized Multiple Intent Conditioned Slot Filling","date":"2023-05-18","arxiv_id":"2305.11023","n_code_links":0,"syntology":null},{"paper":"/paper/generalized-planning-in-pddl-domains-with","slug":"generalized-planning-in-pddl-domains-with","title":"Generalized Planning in PDDL Domains with Pretrained Large Language Models","date":"2023-05-18","arxiv_id":"2305.11014","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tomsilver/llm-genplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-can-be-guided-to-evade","slug":"large-language-models-can-be-guided-to-evade","title":"Large Language Models can be Guided to Evade AI-Generated Text Detection","date":"2023-05-18","arxiv_id":"2305.10847","n_code_links":1,"syntology":null},{"paper":"/paper/coedit-text-editing-by-task-specific","slug":"coedit-text-editing-by-task-specific","title":"CoEdIT: Text Editing by Task-Specific Instruction Tuning","date":"2023-05-17","arxiv_id":"2305.09857","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-chocolate-bunny-to-chocolate-crocodile","title":"From chocolate bunny to chocolate crocodile: Do Language Models Understand Noun Compounds?","date":"2023-05-17","arxiv_id":"2305.10568","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-completion-models-are-few","title":"Knowledge Graph Completion Models are Few-shot Learners: An Empirical Study of Relation Labeling in E-commerce with LLMs","date":"2023-05-17","arxiv_id":"2305.09858","n_code_links":0,"syntology":null},{"paper":"/paper/m3ke-a-massive-multi-level-multi-subject","slug":"m3ke-a-massive-multi-level-multi-subject","title":"M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2023-05-17","arxiv_id":"2305.10263","n_code_links":1,"syntology":null},{"paper":null,"slug":"when-gradient-descent-meets-derivative-free","title":"When Gradient Descent Meets Derivative-Free Optimization: A Match Made in Black-Box Scenario","date":"2023-05-17","arxiv_id":"2305.10013","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-preliminary-analysis-on-the-code-generation","title":"A Preliminary Analysis on the Code Generation Capabilities of GPT-3.5 and Bard AI Models for Java Functions","date":"2023-05-16","arxiv_id":"2305.09402","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-rumination-for-pre-trained-language","slug":"knowledge-rumination-for-pre-trained-language","title":"Knowledge Rumination for Pre-trained Language Models","date":"2023-05-15","arxiv_id":"2305.08732","n_code_links":1,"syntology":null},{"paper":"/paper/rl4f-generating-natural-language-feedback","slug":"rl4f-generating-natural-language-feedback","title":"RL4F: Generating Natural Language Feedback with Reinforcement Learning for Repairing Model Outputs","date":"2023-05-15","arxiv_id":"2305.08844","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["feyzaakyurek/rl4f"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/schema-adaptable-knowledge-graph-construction","slug":"schema-adaptable-knowledge-graph-construction","title":"Schema-adaptable Knowledge Graph Construction","date":"2023-05-15","arxiv_id":"2305.08703","n_code_links":1,"syntology":null},{"paper":"/paper/similarity-weighted-construction-of","slug":"similarity-weighted-construction-of","title":"Similarity-weighted Construction of Contextualized Commonsense Knowledge Graphs for Knowledge-intense Argumentation Tasks","date":"2023-05-15","arxiv_id":"2305.08495","n_code_links":1,"syntology":null},{"paper":"/paper/small-models-are-valuable-plug-ins-for-large","slug":"small-models-are-valuable-plug-ins-for-large","title":"Small Models are Valuable Plug-ins for Large Language Models","date":"2023-05-15","arxiv_id":"2305.08848","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JetRunner/SuperICL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/text-classification-via-large-language-models","slug":"text-classification-via-large-language-models","title":"Text Classification via Large Language Models","date":"2023-05-15","arxiv_id":"2305.08377","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shannonai/gpt-cls-carp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"using-llm-assisted-annotation-for-corpus","title":"Assessing the potential of LLM-assisted annotation for corpus-based pragmatics and discourse analysis: The case of apology","date":"2023-05-15","arxiv_id":"2305.08339","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-emergent-goal-like-behaviour-in","slug":"investigating-emergent-goal-like-behaviour-in","title":"The Machine Psychology of Cooperation: Can GPT models operationalise prompts for altruism, cooperation, competitiveness and selfishness in economic games?","date":"2023-05-13","arxiv_id":"2305.07970","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["phelps-sg/llm-cooperation","gitlab.com/sphelps/llm-cooperation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tinystories-how-small-can-language-models-be","slug":"tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","arxiv_id":"2305.07759","n_code_links":8,"syntology":{"ran":10,"of":18,"n_ran_checked":8,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":null,"slug":"when-giant-language-brains-just-aren-t-enough","title":"When Giant Language Brains Just Aren't Enough! Domain Pizzazz with Knowledge Sparkle Dust","date":"2023-05-12","arxiv_id":"2305.07230","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-can-be-used-to","title":"Spear Phishing With Large Language Models","date":"2023-05-11","arxiv_id":"2305.06972","n_code_links":0,"syntology":null},{"paper":null,"slug":"overinformative-question-answering-by-humans","title":"Overinformative Question Answering by Humans and Machines","date":"2023-05-11","arxiv_id":"2305.07151","n_code_links":0,"syntology":null},{"paper":null,"slug":"recommendation-as-instruction-following-a","title":"Recommendation as Instruction Following: A Large Language Model Empowered Recommendation Approach","date":"2023-05-11","arxiv_id":"2305.07001","n_code_links":0,"syntology":null},{"paper":null,"slug":"bits-of-grass-does-gpt-already-know-how-to","title":"Bits of Grass: Does GPT already know how to write like Whitman?","date":"2023-05-10","arxiv_id":"2305.11064","n_code_links":0,"syntology":null},{"paper":null,"slug":"davinci-the-dualist-the-mind-body-divide-in","title":"Davinci the Dualist: the mind-body divide in large language models and in human learners","date":"2023-05-10","arxiv_id":"2305.07667","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-medically-accurate-summaries-of","title":"Generating medically-accurate summaries of patient-provider dialogue: A multi-stage approach using large language models","date":"2023-05-10","arxiv_id":"2305.05982","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-in-biomedical-natural","slug":"large-language-models-in-biomedical-natural","title":"Benchmarking large language models for biomedical natural language processing applications and recommendations","date":"2023-05-10","arxiv_id":"2305.16326","n_code_links":1,"syntology":null},{"paper":"/paper/summarizing-simplifying-and-synthesizing","slug":"summarizing-simplifying-and-synthesizing","title":"Summarizing, Simplifying, and Synthesizing Medical Evidence Using GPT-3 (with Varying Success)","date":"2023-05-10","arxiv_id":"2305.06299","n_code_links":1,"syntology":null},{"paper":"/paper/codeie-large-code-generation-models-are","slug":"codeie-large-code-generation-models-are","title":"CodeIE: Large Code Generation Models are Better Few-Shot Information Extractors","date":"2023-05-09","arxiv_id":"2305.05711","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["dasepli/codeie"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"towards-an-automatic-optimisation-model","title":"Towards an Automatic Optimisation Model Generator Assisted with Generative Pre-trained Transformer","date":"2023-05-09","arxiv_id":"2305.05811","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-large-language-models-show-decision","title":"Do Large Language Models Show Decision Heuristics Similar to Humans? A Case Study Using GPT-3.5","date":"2023-05-08","arxiv_id":"2305.04400","n_code_links":0,"syntology":null},{"paper":"/paper/explanation-based-finetuning-makes-models","slug":"explanation-based-finetuning-makes-models","title":"Explanation-based Finetuning Makes Models More Robust to Spurious Cues","date":"2023-05-08","arxiv_id":"2305.04990","n_code_links":1,"syntology":null},{"paper":null,"slug":"gersteinlab-at-mediqa-chat-2023-clinical-note","title":"GersteinLab at MEDIQA-Chat 2023: Clinical Note Summarization from Doctor-Patient Conversations through Fine-tuning and In-context Learning","date":"2023-05-08","arxiv_id":"2305.05001","n_code_links":0,"syntology":null},{"paper":"/paper/neurocomparatives-neuro-symbolic-distillation","slug":"neurocomparatives-neuro-symbolic-distillation","title":"NeuroComparatives: Neuro-Symbolic Distillation of Comparative Knowledge","date":"2023-05-08","arxiv_id":"2305.04978","n_code_links":1,"syntology":null}],"record_sha256":"53cfb336060b33ab99786ddfdfc540754b4ff93ba1598b884241e6d15ba91338","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}