{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/13","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":20,"rows_per_page":100,"rows":[1201,1300],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/12","next":"/method/gpt-3/papers/14","papers":[{"paper":"/paper/coupling-large-language-models-with-logic","slug":"coupling-large-language-models-with-logic","title":"Coupling Large Language Models with Logic Programming for Robust and General Reasoning from Text","date":"2023-07-15","arxiv_id":"2307.07696","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azreasoners/llm-asp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-as-superpositions-of","title":"Large Language Models as Superpositions of Cultural Perspectives","date":"2023-07-15","arxiv_id":"2307.07870","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-large-language-models-to-generate","slug":"leveraging-large-language-models-to-generate","title":"Leveraging Large Language Models to Generate Answer Set Programs","date":"2023-07-15","arxiv_id":"2307.07699","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["azreasoners/gpt-asp-rules"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-study-on-differentiable-logic-and-llms-for","title":"A Study on Differentiable Logic and LLMs for EPIC-KITCHENS-100 Unsupervised Domain Adaptation Challenge for Action Recognition 2023","date":"2023-07-13","arxiv_id":"2307.06569","n_code_links":0,"syntology":null},{"paper":null,"slug":"agreement-tracking-for-multi-issue","title":"Agreement Tracking for Multi-Issue Negotiation Dialogues","date":"2023-07-13","arxiv_id":"2307.06524","n_code_links":0,"syntology":null},{"paper":"/paper/negated-complementary-commonsense-using-large","slug":"negated-complementary-commonsense-using-large","title":"Negated Complementary Commonsense using Large Language Models","date":"2023-07-13","arxiv_id":"2307.06794","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for","title":"Distilling Large Language Models for Biomedical Knowledge Extraction: A Case Study on Adverse Drug Events","date":"2023-07-12","arxiv_id":"2307.06439","n_code_links":0,"syntology":null},{"paper":null,"slug":"argumentative-segmentation-enhancement-for","title":"Argumentative Segmentation Enhancement for Legal Summarization","date":"2023-07-11","arxiv_id":"2307.05081","n_code_links":0,"syntology":null},{"paper":"/paper/unleashing-cognitive-synergy-in-large","slug":"unleashing-cognitive-synergy-in-large","title":"Unleashing the Emergent Cognitive Synergy in Large Language Models: A Task-Solving Agent through Multi-Persona Self-Collaboration","date":"2023-07-11","arxiv_id":"2307.05300","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-stitch-in-time-saves-nine-detecting-and","title":"A Stitch in Time Saves Nine: Detecting and Mitigating Hallucinations of LLMs by Validating Low-Confidence Generation","date":"2023-07-08","arxiv_id":"2307.03987","n_code_links":0,"syntology":null},{"paper":"/paper/dwreco-at-checkthat-2023-enhancing","slug":"dwreco-at-checkthat-2023-enhancing","title":"DWReCO at CheckThat! 2023: Enhancing Subjectivity Detection through Style-based Data Sampling","date":"2023-07-07","arxiv_id":"2307.03550","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-does-ai-chat-change-search-behaviors","title":"How does AI chat change search behaviors?","date":"2023-07-07","arxiv_id":"2307.03826","n_code_links":0,"syntology":null},{"paper":null,"slug":"radar-robust-ai-text-detection-via","title":"RADAR: Robust AI-Text Detection via Adversarial Learning","date":"2023-07-07","arxiv_id":"2307.03838","n_code_links":0,"syntology":null},{"paper":"/paper/improving-retrieval-augmented-large-language","slug":"improving-retrieval-augmented-large-language","title":"Improving Retrieval-Augmented Large Language Models via Data Importance Learning","date":"2023-07-06","arxiv_id":"2307.03027","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["amsterdata/ragbooster"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/text-alignment-is-an-efficient-unified-model","slug":"text-alignment-is-an-efficient-unified-model","title":"Text Alignment Is An Efficient Unified Model for Massive NLP Tasks","date":"2023-07-06","arxiv_id":"2307.02729","n_code_links":1,"syntology":null},{"paper":"/paper/external-reasoning-towards-multi-large","slug":"external-reasoning-towards-multi-large","title":"External Reasoning: Towards Multi-Large-Language-Models Interchangeable Assistance with Human Feedback","date":"2023-07-05","arxiv_id":"2307.12057","n_code_links":1,"syntology":null},{"paper":"/paper/hoodwinked-deception-and-cooperation-in-a","slug":"hoodwinked-deception-and-cooperation-in-a","title":"Hoodwinked: Deception and Cooperation in a Text-Based Game for Language Models","date":"2023-07-05","arxiv_id":"2308.01404","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-controllable-transformer-based","slug":"multilingual-controllable-transformer-based","title":"Multilingual Controllable Transformer-Based Lexical Simplification","date":"2023-07-05","arxiv_id":"2307.02120","n_code_links":1,"syntology":null},{"paper":null,"slug":"open-source-large-language-models-outperform","title":"Open-Source LLMs for Text Annotation: A Practical Guide for Model Setting and Fine-Tuning","date":"2023-07-05","arxiv_id":"2307.02179","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-formai-dataset-generative-ai-in-software","title":"The FormAI Dataset: Generative AI in Software Security Through the Lens of Formal Verification","date":"2023-07-05","arxiv_id":"2307.02192","n_code_links":0,"syntology":null},{"paper":"/paper/embodied-task-planning-with-large-language","slug":"embodied-task-planning-with-large-language","title":"Embodied Task Planning with Large Language Models","date":"2023-07-04","arxiv_id":"2307.01848","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Gary3410/TaPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/kdstm-neural-semi-supervised-topic-modeling","slug":"kdstm-neural-semi-supervised-topic-modeling","title":"KDSTM: Neural Semi-supervised Topic Modeling with Knowledge Distillation","date":"2023-07-04","arxiv_id":"2307.01878","n_code_links":0,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"interpretability-and-transparency-driven","title":"Interpretability and Transparency-Driven Detection and Transformation of Textual Adversarial Examples (IT-DT)","date":"2023-07-03","arxiv_id":"2307.01225","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterative-zero-shot-llm-prompting-for","title":"Iterative Zero-Shot LLM Prompting for Knowledge Graph Construction","date":"2023-07-03","arxiv_id":"2307.01128","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-gpt-for-automating","title":"Large Language Models (GPT) for automating feedback on programming assignments","date":"2023-06-30","arxiv_id":"2307.00150","n_code_links":0,"syntology":null},{"paper":"/paper/meta-reasoning-semantics-symbol","slug":"meta-reasoning-semantics-symbol","title":"Meta-Reasoning: Semantics-Symbol Deconstruction for Large Language Models","date":"2023-06-30","arxiv_id":"2306.17820","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-negation-detection-assessment-of-gpts","title":"A negation detection assessment of GPTs: analysis with the xNot360 dataset","date":"2023-06-29","arxiv_id":"2306.16638","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-large-language-model","title":"Benchmarking Large Language Model Capabilities for Conditional Generation","date":"2023-06-29","arxiv_id":"2306.16793","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-for-programming-education","title":"Generative AI for Programming Education: Benchmarking ChatGPT, GPT-4, and Human Tutors","date":"2023-06-29","arxiv_id":"2306.17156","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-calibration-and-error-correction","title":"Pareto Optimal Learning for Estimating Large Language Model Errors","date":"2023-06-28","arxiv_id":"2306.16564","n_code_links":0,"syntology":null},{"paper":null,"slug":"inferring-the-goals-of-communicating-agents","title":"Inferring the Goals of Communicating Agents from Actions and Instructions","date":"2023-06-28","arxiv_id":"2306.16207","n_code_links":0,"syntology":null},{"paper":"/paper/is-chatgpt-a-biomedical-expert-exploring-the","slug":"is-chatgpt-a-biomedical-expert-exploring-the","title":"Is ChatGPT a Biomedical Expert? -- Exploring the Zero-Shot Performance of Current GPT Models in Biomedical Tasks","date":"2023-06-28","arxiv_id":"2306.16108","n_code_links":1,"syntology":null},{"paper":"/paper/taqyim-evaluating-arabic-nlp-tasks-using","slug":"taqyim-evaluating-arabic-nlp-tasks-using","title":"Taqyim: Evaluating Arabic NLP Tasks Using ChatGPT Models","date":"2023-06-28","arxiv_id":"2306.16322","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-gpt-3-5-and-gpt-4-on-grammatical","title":"Evaluating GPT-3.5 and GPT-4 on Grammatical Error Correction for Brazilian Portuguese","date":"2023-06-27","arxiv_id":"2306.15788","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-robustness-of-large-language","title":"Exploring the Robustness of Large Language Models for Solving Programming Problems","date":"2023-06-26","arxiv_id":"2306.14583","n_code_links":0,"syntology":null},{"paper":null,"slug":"let-s-do-a-thought-experiment-using","title":"Let's Do a Thought Experiment: Using Counterfactuals to Improve Moral Reasoning","date":"2023-06-25","arxiv_id":"2306.14308","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-as-sous-chefs-revising","slug":"large-language-models-as-sous-chefs-revising","title":"Large Language Models as Sous Chefs: Revising Recipes with GPT-3","date":"2023-06-24","arxiv_id":"2306.13986","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-sequence-models-for-sequential-decision","title":"Large Sequence Models for Sequential Decision-Making: A Survey","date":"2023-06-24","arxiv_id":"2306.13945","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-uses-of-large-language-models-to","title":"On the Uses of Large Language Models to Interpret Ambiguous Cyberattack Descriptions","date":"2023-06-24","arxiv_id":"2306.14062","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-assisted-content-analysis-using-large","title":"LLM-Assisted Content Analysis: Using Large Language Models to Support Deductive Coding","date":"2023-06-23","arxiv_id":"2306.14924","n_code_links":0,"syntology":null},{"paper":"/paper/system-level-natural-language-feedback","slug":"system-level-natural-language-feedback","title":"System-Level Natural Language Feedback","date":"2023-06-23","arxiv_id":"2306.13588","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yyy-apple/sys-nl-feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/cross-lingual-cross-temporal-summarization","slug":"cross-lingual-cross-temporal-summarization","title":"Cross-lingual Cross-temporal Summarization: Dataset, Models, Evaluation","date":"2023-06-22","arxiv_id":"2306.12916","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-to-gpt-3-step-by-step-thinking","slug":"prompt-to-gpt-3-step-by-step-thinking","title":"Prompt to GPT-3: Step-by-Step Thinking Instructions for Humor Generation","date":"2023-06-22","arxiv_id":"2306.13195","n_code_links":1,"syntology":null},{"paper":"/paper/solving-and-generating-npr-sunday-puzzles","slug":"solving-and-generating-npr-sunday-puzzles","title":"Solving and Generating NPR Sunday Puzzles with Large Language Models","date":"2023-06-21","arxiv_id":"2306.12255","n_code_links":1,"syntology":null},{"paper":null,"slug":"which-spurious-correlations-impact-reasoning","title":"Which Spurious Correlations Impact Reasoning in NLI Models? A Visual Interactive Diagnosis through Data-Constrained Counterfactuals","date":"2023-06-21","arxiv_id":"2306.12146","n_code_links":0,"syntology":null},{"paper":null,"slug":"decodingtrust-a-comprehensive-assessment-of","title":"DecodingTrust: A Comprehensive Assessment of Trustworthiness in GPT Models","date":"2023-06-20","arxiv_id":"2306.11698","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","n_code_links":1,"syntology":null},{"paper":null,"slug":"textbooks-are-all-you-need","title":"Textbooks Are All You Need","date":"2023-06-20","arxiv_id":"2306.11644","n_code_links":0,"syntology":null},{"paper":"/paper/a-preliminary-study-of-chatgpt-on-news","slug":"a-preliminary-study-of-chatgpt-on-news","title":"A Preliminary Study of ChatGPT on News Recommendation: Personalization, Provider Fairness, Fake News","date":"2023-06-19","arxiv_id":"2306.10702","n_code_links":1,"syntology":null},{"paper":"/paper/bayling-bridging-cross-lingual-alignment-and","slug":"bayling-bridging-cross-lingual-alignment-and","title":"BayLing: Bridging Cross-lingual Alignment and Instruction Following through Interactive Translation for Large Language Models","date":"2023-06-19","arxiv_id":"2306.10968","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-social-network-hate-detection-using","slug":"enhancing-social-network-hate-detection-using","title":"Enhancing social network hate detection using back translation and GPT-3 augmentations during training and test-time","date":"2023-06-17","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/demystifying-gpt-self-repair-for-code","slug":"demystifying-gpt-self-repair-for-code","title":"Is Self-Repair a Silver Bullet for Code Generation?","date":"2023-06-16","arxiv_id":"2306.09896","n_code_links":1,"syntology":null},{"paper":"/paper/explore-establish-exploit-red-teaming","slug":"explore-establish-exploit-red-teaming","title":"Explore, Establish, Exploit: Red Teaming Language Models from Scratch","date":"2023-06-15","arxiv_id":"2306.09442","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["algorithmic-alignment-lab/commonclaim","thestephencasper/common_claim","thestephencasper/explore_establish_exploit_llms"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"exploring-the-mit-mathematics-and-eecs","title":"Exploring the MIT Mathematics and EECS Curriculum Using Large Language Models","date":"2023-06-15","arxiv_id":"2306.08997","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-the-effectiveness-of-gpt-3-in","slug":"assessing-the-effectiveness-of-gpt-3-in","title":"Assessing the Effectiveness of GPT-3 in Detecting False Political Statements: A Case Study on the LIAR Dataset","date":"2023-06-14","arxiv_id":"2306.08190","n_code_links":1,"syntology":null},{"paper":"/paper/language-models-are-not-naysayers-an-analysis","slug":"language-models-are-not-naysayers-an-analysis","title":"Language models are not naysayers: An analysis of language models on negation benchmarks","date":"2023-06-14","arxiv_id":"2306.08189","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-social-network-hate-detection-using-1","slug":"enhancing-social-network-hate-detection-using-1","title":"Enhancing Social Network Hate Detection Using Back Translation and GPT-3 Augmentations During Training and Test-Time","date":"2023-06-13","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"flame-few-shot-learning-from-natural-language","title":"FLamE: Few-shot Learning from Natural Language Explanations","date":"2023-06-13","arxiv_id":"2306.08042","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-like-intuitive-behavior-and-reasoning","title":"Human-Like Intuitive Behavior and Reasoning Biases Emerged in Language Models -- and Disappeared in GPT-4","date":"2023-06-13","arxiv_id":"2306.07622","n_code_links":0,"syntology":null},{"paper":"/paper/recursion-of-thought-a-divide-and-conquer","slug":"recursion-of-thought-a-divide-and-conquer","title":"Recursion of Thought: A Divide-and-Conquer Approach to Multi-Context Reasoning with Language Models","date":"2023-06-12","arxiv_id":"2306.06891","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-bea-2023-shared-task-on-generating-ai","title":"The BEA 2023 Shared Task on Generating AI Teacher Responses in Educational Dialogues","date":"2023-06-12","arxiv_id":"2306.06941","n_code_links":0,"syntology":null},{"paper":"/paper/waffling-around-for-performance-visual","slug":"waffling-around-for-performance-visual","title":"Waffling around for Performance: Visual Classification with Random Words and Broad Concepts","date":"2023-06-12","arxiv_id":"2306.07282","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["explainableml/waffleclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/inductive-reasoning-in-humans-and-large","slug":"inductive-reasoning-in-humans-and-large","title":"Inductive reasoning in humans and large language models","date":"2023-06-11","arxiv_id":"2306.06548","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-responses-of-large-language","title":"Exploring the Responses of Large Language Models to Beginner Programmers' Help Requests","date":"2023-06-09","arxiv_id":"2306.05715","n_code_links":0,"syntology":null},{"paper":"/paper/reliability-check-an-analysis-of-gpt-3-s","slug":"reliability-check-an-analysis-of-gpt-3-s","title":"Reliability Check: An Analysis of GPT-3's Response to Sensitive Topics and Prompt Wording","date":"2023-06-09","arxiv_id":"2306.06199","n_code_links":2,"syntology":null},{"paper":"/paper/pandalm-an-automatic-evaluation-benchmark-for","slug":"pandalm-an-automatic-evaluation-benchmark-for","title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","date":"2023-06-08","arxiv_id":"2306.05087","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["weopenml/pandalm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/prefer-to-classify-improving-text-classifiers","slug":"prefer-to-classify-improving-text-classifiers","title":"Prefer to Classify: Improving Text Classifiers via Auxiliary Preference Learning","date":"2023-06-08","arxiv_id":"2306.04925","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["minnesotanlp/p2c"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-adaio-system-at-the-bea-2023-shared-task","title":"The ADAIO System at the BEA-2023 Shared Task on Generating AI Teacher Responses in Educational Dialogues","date":"2023-06-08","arxiv_id":"2306.05360","n_code_links":0,"syntology":null},{"paper":"/paper/toolalpaca-generalized-tool-learning-for","slug":"toolalpaca-generalized-tool-learning-for","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","date":"2023-06-08","arxiv_id":"2306.05301","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["tangqiaoyu/ToolAlpaca"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/good-data-large-data-or-no-data-comparing","slug":"good-data-large-data-or-no-data-comparing","title":"Good Data, Large Data, or No Data? Comparing Three Approaches in Developing Research Aspect Classifiers for Biomedical Papers","date":"2023-06-07","arxiv_id":"2306.04820","n_code_links":1,"syntology":null},{"paper":null,"slug":"personality-testing-of-gpt-3-limited-temporal","title":"Personality testing of Large Language Models: Limited temporal stability, but highlighted prosociality","date":"2023-06-07","arxiv_id":"2306.04308","n_code_links":0,"syntology":null},{"paper":null,"slug":"sciencebenchmark-a-complex-real-world","title":"ScienceBenchmark: A Complex Real-World Benchmark for Evaluating Natural Language to SQL Systems","date":"2023-06-07","arxiv_id":"2306.04743","n_code_links":0,"syntology":null},{"paper":"/paper/the-two-word-test-a-semantic-benchmark-for","slug":"the-two-word-test-a-semantic-benchmark-for","title":"The Two Word Test: A Semantic Benchmark for Large Language Models","date":"2023-06-07","arxiv_id":"2306.04610","n_code_links":1,"syntology":null},{"paper":"/paper/certified-reasoning-with-language-models","slug":"certified-reasoning-with-language-models","title":"Certified Deductive Reasoning with Language Models","date":"2023-06-06","arxiv_id":"2306.04031","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"iterative-translation-refinement-with-large","title":"Iterative Translation Refinement with Large Language Models","date":"2023-06-06","arxiv_id":"2306.03856","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-syntactic-generalization-capacity","title":"Analyzing Syntactic Generalization Capacity of Pre-trained Language Models on Japanese Honorific Conversion","date":"2023-06-05","arxiv_id":"2306.03055","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-as-a-mapping-assistant-a-novel-method","title":"ChatGPT as a mapping assistant: A novel method to enrich maps with generative AI and content derived from street-level photographs","date":"2023-06-05","arxiv_id":"2306.03204","n_code_links":0,"syntology":null},{"paper":"/paper/auto-gpt-for-online-decision-making","slug":"auto-gpt-for-online-decision-making","title":"Auto-GPT for Online Decision Making: Benchmarks and Additional Opinions","date":"2023-06-04","arxiv_id":"2306.02224","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["younghuman/llmagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-coding-social-science-datasets-with-1","title":"Towards Coding Social Science Datasets with Language Models","date":"2023-06-03","arxiv_id":"2306.02177","n_code_links":0,"syntology":null},{"paper":"/paper/multi-dimensional-evaluation-of-text","slug":"multi-dimensional-evaluation-of-text","title":"Multi-Dimensional Evaluation of Text Summarization with In-Context Learning","date":"2023-06-01","arxiv_id":"2306.01200","n_code_links":1,"syntology":null},{"paper":null,"slug":"systematic-evaluation-of-gpt-3-for-zero-shot","title":"Systematic Evaluation of GPT-3 for Zero-Shot Personality Estimation","date":"2023-06-01","arxiv_id":"2306.01183","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-gpt-s-programming-capability","title":"Evaluating GPT's Programming Capability through CodeWars' Katas","date":"2023-05-31","arxiv_id":"2306.01784","n_code_links":0,"syntology":null},{"paper":null,"slug":"examining-the-emergence-of-deductive","title":"Examining the Emergence of Deductive Reasoning in Generative Language Models","date":"2023-05-31","arxiv_id":"2306.01009","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-base-question-answering-for-space","slug":"knowledge-base-question-answering-for-space","title":"Knowledge Base Question Answering for Space Debris Queries","date":"2023-05-31","arxiv_id":"2305.19734","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-conceptual-representation-require","title":"Does Conceptual Representation Require Embodiment? Insights From Large Language Models","date":"2023-05-30","arxiv_id":"2305.19103","n_code_links":0,"syntology":null},{"paper":null,"slug":"generate-then-select-open-ended-visual","title":"Generate then Select: Open-ended Visual Question Answering Guided by World Knowledge","date":"2023-05-30","arxiv_id":"2305.18842","n_code_links":0,"syntology":null},{"paper":"/paper/check-covid-fact-checking-covid-19-news","slug":"check-covid-fact-checking-covid-19-news","title":"Check-COVID: Fact-Checking COVID-19 News Claims with Scientific Evidence","date":"2023-05-29","arxiv_id":"2305.18265","n_code_links":1,"syntology":null},{"paper":null,"slug":"coeditor-leveraging-contextual-changes-for","title":"Coeditor: Leveraging Contextual Changes for Multi-round Code Auto-editing","date":"2023-05-29","arxiv_id":"2305.18584","n_code_links":0,"syntology":null},{"paper":"/paper/do-large-language-models-know-what-they-don-t","slug":"do-large-language-models-know-what-they-don-t","title":"Do Large Language Models Know What They Don't Know?","date":"2023-05-29","arxiv_id":"2305.18153","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinzhangyue/selfaware"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-effectiveness-of-gpt-3-in","title":"Exploring Effectiveness of GPT-3 in Grammatical Error Correction: A Study on Performance and Controllability in Prompt-Based Methods","date":"2023-05-29","arxiv_id":"2305.18156","n_code_links":0,"syntology":null},{"paper":"/paper/lm-cppf-paraphrasing-guided-data-augmentation","slug":"lm-cppf-paraphrasing-guided-data-augmentation","title":"LM-CPPF: Paraphrasing-Guided Data Augmentation for Contrastive Prompt-Based Few-Shot Fine-Tuning","date":"2023-05-29","arxiv_id":"2305.18169","n_code_links":1,"syntology":null},{"paper":"/paper/marked-personas-using-natural-language","slug":"marked-personas-using-natural-language","title":"Marked Personas: Using Natural Language Prompts to Measure Stereotypes in Language Models","date":"2023-05-29","arxiv_id":"2305.18189","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["myracheng/markedpersonas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/syntax-and-semantics-meet-in-the-middle","slug":"syntax-and-semantics-meet-in-the-middle","title":"Syntax and Semantics Meet in the \"Middle\": Probing the Syntax-Semantics Interface of LMs Through Agentivity","date":"2023-05-29","arxiv_id":"2305.18185","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-gpt-3-generated-explanations-for","slug":"evaluating-gpt-3-generated-explanations-for","title":"Evaluating GPT-3 Generated Explanations for Hateful Content Moderation","date":"2023-05-28","arxiv_id":"2305.17680","n_code_links":1,"syntology":null},{"paper":"/paper/generating-edu-extracts-for-plan-guided","slug":"generating-edu-extracts-for-plan-guided","title":"Generating EDU Extracts for Plan-Guided Summary Re-Ranking","date":"2023-05-28","arxiv_id":"2305.17779","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":0,"n_instrument":6,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["griff4692/edu-sum"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/kosbi-a-dataset-for-mitigating-social-bias","slug":"kosbi-a-dataset-for-mitigating-social-bias","title":"KoSBi: A Dataset for Mitigating Social Bias Risks Towards Safer Large Language Model Application","date":"2023-05-28","arxiv_id":"2305.17701","n_code_links":1,"syntology":null},{"paper":"/paper/mitigating-label-biases-for-in-context","slug":"mitigating-label-biases-for-in-context","title":"Mitigating Label Biases for In-context Learning","date":"2023-05-28","arxiv_id":"2305.19148","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fywalter/label-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/square-a-large-scale-dataset-of-sensitive","slug":"square-a-large-scale-dataset-of-sensitive","title":"SQuARe: A Large-Scale Dataset of Sensitive Questions and Acceptable Responses Created Through Human-Machine Collaboration","date":"2023-05-28","arxiv_id":"2305.17696","n_code_links":1,"syntology":null},{"paper":null,"slug":"complementary-and-integrative-health-lexicon","title":"Complementary and Integrative Health Lexicon (CIHLex) and Entity Recognition in the Literature","date":"2023-05-27","arxiv_id":"2305.17353","n_code_links":0,"syntology":null},{"paper":"/paper/dna-gpt-divergent-n-gram-analysis-for","slug":"dna-gpt-divergent-n-gram-analysis-for","title":"DNA-GPT: Divergent N-Gram Analysis for Training-Free Detection of GPT-Generated Text","date":"2023-05-27","arxiv_id":"2305.17359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xianjun-yang/dna-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"af17d5b84309eae8e5073c95b59e50faae675249c8b301cffe36f43c31d84b04","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}