{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/11","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":20,"rows_per_page":100,"rows":[1001,1100],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/10","next":"/method/gpt-3/papers/12","papers":[{"paper":"/paper/entity-matching-using-large-language-models","slug":"entity-matching-using-large-language-models","title":"Entity Matching using Large Language Models","date":"2023-10-17","arxiv_id":"2310.11244","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-llms-for-privilege-escalation","slug":"evaluating-llms-for-privilege-escalation","title":"LLMs as Hackers: Autonomous Linux Privilege Escalation Attacks","date":"2023-10-17","arxiv_id":"2310.11409","n_code_links":1,"syntology":null},{"paper":"/paper/intent-detection-and-slot-filling-for-home","slug":"intent-detection-and-slot-filling-for-home","title":"Intent Detection and Slot Filling for Home Assistants: Dataset and Analysis for Bangla and Sylheti","date":"2023-10-17","arxiv_id":"2310.10935","n_code_links":1,"syntology":null},{"paper":"/paper/probing-the-creativity-of-large-language","slug":"probing-the-creativity-of-large-language","title":"Probing the Creativity of Large Language Models: Can models produce divergent semantic association?","date":"2023-10-17","arxiv_id":"2310.11158","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dingnlab/probing_creativity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"utilising-a-large-language-model-to-annotate","title":"Utilising a Large Language Model to Annotate Subject Metadata: A Case Study in an Australian National Research Data Catalogue","date":"2023-10-17","arxiv_id":"2310.11318","n_code_links":0,"syntology":null},{"paper":null,"slug":"battle-of-the-large-language-models-dolly-vs","title":"Battle of the Large Language Models: Dolly vs LLaMA vs Vicuna vs Guanaco vs Bard vs ChatGPT -- A Text-to-SQL Parsing Comparison","date":"2023-10-16","arxiv_id":"2310.10190","n_code_links":0,"syntology":null},{"paper":"/paper/bioplanner-automatic-evaluation-of-llms-on","slug":"bioplanner-automatic-evaluation-of-llms-on","title":"BioPlanner: Automatic Evaluation of LLMs on Protocol Planning in Biology","date":"2023-10-16","arxiv_id":"2310.10632","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bioplanner/bioplanner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fine-tuning-chatgpt-for-automatic-scoring","title":"Fine-tuning ChatGPT for Automatic Scoring","date":"2023-10-16","arxiv_id":"2310.10072","n_code_links":0,"syntology":null},{"paper":null,"slug":"prediction-of-arabic-legal-rulings-using","title":"Prediction of Arabic Legal Rulings using Large Language Models","date":"2023-10-16","arxiv_id":"2310.10260","n_code_links":0,"syntology":null},{"paper":"/paper/transom-an-efficient-fault-tolerant-system","slug":"transom-an-efficient-fault-tolerant-system","title":"TRANSOM: An Efficient Fault-Tolerant System for Training LLMs","date":"2023-10-16","arxiv_id":"2310.10046","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["SenseCore/transom-checkpoint-engine"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"large-language-model-aware-in-context","title":"Large Language Model-Aware In-Context Learning for Code Generation","date":"2023-10-15","arxiv_id":"2310.09748","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-in-context-student","slug":"large-language-models-for-in-context-student","title":"Large Language Models for In-Context Student Modeling: Synthesizing Student's Behavior in Visual Programming","date":"2023-10-15","arxiv_id":"2310.10690","n_code_links":1,"syntology":null},{"paper":"/paper/a-systematic-evaluation-of-large-language-1","slug":"a-systematic-evaluation-of-large-language-1","title":"Assessing and Enhancing the Robustness of Large Language Models with Task Structure Variations for Logical Reasoning","date":"2023-10-13","arxiv_id":"2310.09430","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["strong-ai-lab/logical-and-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/human-in-the-loop-machine-translation-with","slug":"human-in-the-loop-machine-translation-with","title":"Human-in-the-loop Machine Translation with Large Language Model","date":"2023-10-13","arxiv_id":"2310.08908","n_code_links":1,"syntology":null},{"paper":null,"slug":"table-gpt-table-tuned-gpt-for-diverse-table","title":"Table-GPT: Table-tuned GPT for Diverse Table Tasks","date":"2023-10-13","arxiv_id":"2310.09263","n_code_links":0,"syntology":null},{"paper":"/paper/jailbreaking-black-box-large-language-models","slug":"jailbreaking-black-box-large-language-models","title":"Jailbreaking Black Box Large Language Models in Twenty Queries","date":"2023-10-12","arxiv_id":"2310.08419","n_code_links":1,"syntology":{"ran":2,"of":7,"n_ran_checked":1,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["patrickrchao/jailbreakingllms"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-can-replicate-cross","title":"Large language models can replicate cross-cultural differences in personality","date":"2023-10-12","arxiv_id":"2310.10679","n_code_links":0,"syntology":null},{"paper":null,"slug":"promptor-a-conversational-and-autonomous","title":"Promptor: A Conversational and Autonomous Prompt Generation Agent for Intelligent Text Entry Techniques","date":"2023-10-12","arxiv_id":"2310.08101","n_code_links":0,"syntology":null},{"paper":"/paper/qasina-religious-domain-question-answering","slug":"qasina-religious-domain-question-answering","title":"QASiNa: Religious Domain Question Answering using Sirah Nabawiyah","date":"2023-10-12","arxiv_id":"2310.08102","n_code_links":1,"syntology":null},{"paper":null,"slug":"diversity-of-thought-improves-reasoning","title":"Diversity of Thought Improves Reasoning Abilities of LLMs","date":"2023-10-11","arxiv_id":"2310.07088","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-landscape-of-large-language","title":"Do Large Language Models have Shared Weaknesses in Medical Question Answering?","date":"2023-10-11","arxiv_id":"2310.07225","n_code_links":0,"syntology":null},{"paper":"/paper/found-in-the-middle-permutation-self","slug":"found-in-the-middle-permutation-self","title":"Found in the Middle: Permutation Self-Consistency Improves Listwise Ranking in Large Language Models","date":"2023-10-11","arxiv_id":"2310.07712","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["castorini/perm-sc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-are-zero-shot-time-1","slug":"large-language-models-are-zero-shot-time-1","title":"Large Language Models Are Zero-Shot Time Series Forecasters","date":"2023-10-11","arxiv_id":"2310.07820","n_code_links":2,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ngruver/llmtime"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"wigenai-the-symphony-of-wireless-and","title":"Diffusion Models for Wireless Communications","date":"2023-10-11","arxiv_id":"2310.07312","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-clinical-coding-using-off-the-shelf","title":"Automated clinical coding using off-the-shelf large language models","date":"2023-10-10","arxiv_id":"2310.06552","n_code_links":0,"syntology":null},{"paper":"/paper/geollm-extracting-geospatial-knowledge-from","slug":"geollm-extracting-geospatial-knowledge-from","title":"GeoLLM: Extracting Geospatial Knowledge from Large Language Models","date":"2023-10-10","arxiv_id":"2310.06213","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rohinmanvi/GeoLLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-for-propaganda","slug":"large-language-models-for-propaganda","title":"Large Language Models for Propaganda Detection","date":"2023-10-10","arxiv_id":"2310.06422","n_code_links":2,"syntology":null},{"paper":"/paper/longllmlingua-accelerating-and-enhancing-llms","slug":"longllmlingua-accelerating-and-enhancing-llms","title":"LongLLMLingua: Accelerating and Enhancing LLMs in Long Context Scenarios via Prompt Compression","date":"2023-10-10","arxiv_id":"2310.06839","n_code_links":3,"syntology":null},{"paper":null,"slug":"cabbage-sweeter-than-cake-analysing-the","title":"Cabbage Sweeter than Cake? Analysing the Potential of Large Language Models for Learning Conceptual Spaces","date":"2023-10-09","arxiv_id":"2310.05481","n_code_links":0,"syntology":null},{"paper":"/paper/mbbc-exploring-the-multilingual-maze","slug":"mbbc-exploring-the-multilingual-maze","title":"Exploring the Maze of Multilingual Modeling","date":"2023-10-09","arxiv_id":"2310.05404","n_code_links":0,"syntology":null},{"paper":null,"slug":"sc-safety-a-multi-round-open-ended-question","title":"SC-Safety: A Multi-round Open-ended Question Adversarial Safety Benchmark for Large Language Models in Chinese","date":"2023-10-09","arxiv_id":"2310.05818","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-program-testing-ability-of-large-language","title":"The Program Testing Ability of Large Language Models for Code","date":"2023-10-09","arxiv_id":"2310.05727","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-emily-and-greg-still-more-employable-than","title":"Are Emily and Greg Still More Employable than Lakisha and Jamal? Investigating Algorithmic Hiring Bias in the Era of ChatGPT","date":"2023-10-08","arxiv_id":"2310.05135","n_code_links":0,"syntology":null},{"paper":"/paper/llm4vv-developing-llm-driven-testsuite-for","slug":"llm4vv-developing-llm-driven-testsuite-for","title":"LLM4VV: Developing LLM-Driven Testsuite for Compiler Validation","date":"2023-10-08","arxiv_id":"2310.04963","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-detection-of-machine-generated","slug":"zero-shot-detection-of-machine-generated","title":"Zero-Shot Detection of Machine-Generated Codes","date":"2023-10-08","arxiv_id":"2310.05103","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-only-pass-primary","slug":"large-language-models-only-pass-primary","title":"Large Language Models Only Pass Primary School Exams in Indonesia: A Comprehensive Test on IndoMMLU","date":"2023-10-07","arxiv_id":"2310.04928","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":1,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["fajri91/indommlu"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/agent-instructs-large-language-models-to-be","slug":"agent-instructs-large-language-models-to-be","title":"Agent Instructs Large Language Models to be General Zero-Shot Reasoners","date":"2023-10-05","arxiv_id":"2310.03710","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wang-research-lab/agentinstruct"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automating-human-tutor-style-programming","slug":"automating-human-tutor-style-programming","title":"Automating Human Tutor-Style Programming Feedback: Leveraging GPT-4 Tutor Model for Hint Generation and GPT-3.5 Student Model for Hint Validation","date":"2023-10-05","arxiv_id":"2310.03780","n_code_links":2,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["machine-teaching-group/lak2024_gpt4-hints-gpt3.5val","machine-teaching-group/lak2024_gpt4hints-gpt3.5val"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dspy-compiling-declarative-language-model","slug":"dspy-compiling-declarative-language-model","title":"DSPy: Compiling Declarative Language Model Calls into Self-Improving Pipelines","date":"2023-10-05","arxiv_id":"2310.03714","n_code_links":3,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["stanfordnlp/dspy"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/fine-tuning-aligned-language-models","slug":"fine-tuning-aligned-language-models","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","date":"2023-10-05","arxiv_id":"2310.03693","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["llm-tuning-safety/llms-finetuning-safety"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"a-survey-of-gpt-3-family-large-language","title":"A Survey of GPT-3 Family Large Language Models Including ChatGPT and GPT-4","date":"2023-10-04","arxiv_id":"2310.12321","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-model-cascades-with-mixture-of","slug":"large-language-model-cascades-with-mixture-of","title":"Large Language Model Cascades with Mixture of Thoughts Representations for Cost-efficient Reasoning","date":"2023-10-04","arxiv_id":"2310.03094","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieval-meets-long-context-large-language","title":"Retrieval meets Long Context Large Language Models","date":"2023-10-04","arxiv_id":"2310.03025","n_code_links":0,"syntology":null},{"paper":"/paper/instance-needs-more-care-rewriting-prompts","slug":"instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","arxiv_id":"2310.02107","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salokr/propmted"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-driver-learning-to-drive-with-gpt","slug":"gpt-driver-learning-to-drive-with-gpt","title":"GPT-Driver: Learning to Drive with GPT","date":"2023-10-02","arxiv_id":"2310.01415","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pointscoder/gpt-driver"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-lies-hallucinations-are-not-bugs-but","slug":"llm-lies-hallucinations-are-not-bugs-but","title":"LLM Lies: Hallucinations are not Bugs, but Features as Adversarial Examples","date":"2023-10-02","arxiv_id":"2310.01469","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["pku-yuangroup/hallucination-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/booookscore-a-systematic-exploration-of-book","slug":"booookscore-a-systematic-exploration-of-book","title":"BooookScore: A systematic exploration of book-length summarization in the era of LLMs","date":"2023-10-01","arxiv_id":"2310.00785","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lilakk/booookscore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gaze-driven-sentence-simplification-for","title":"Gaze-Driven Sentence Simplification for Language Learners: Enhancing Comprehension and Readability","date":"2023-09-30","arxiv_id":"2310.00355","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-large-language-model-approach-to","title":"A Large Language Model Approach to Educational Survey Feedback Analysis","date":"2023-09-29","arxiv_id":"2309.17447","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-the-abilities-of-large-language","slug":"benchmarking-the-abilities-of-large-language","title":"Benchmarking the Abilities of Large Language Models for RDF Knowledge Graph Creation and Comprehension: How Well Do LLMs Speak Turtle?","date":"2023-09-29","arxiv_id":"2309.17122","n_code_links":3,"syntology":null},{"paper":"/paper/dyval-graph-informed-dynamic-evaluation-of","slug":"dyval-graph-informed-dynamic-evaluation-of","title":"DyVal: Dynamic Evaluation of Large Language Models for Reasoning Tasks","date":"2023-09-29","arxiv_id":"2309.17167","n_code_links":1,"syntology":null},{"paper":"/paper/llm-deliberation-evaluating-llms-with","slug":"llm-deliberation-evaluating-llms-with","title":"Cooperation, Competition, and Maliciousness: LLM-Stakeholders Interactive Negotiation","date":"2023-09-29","arxiv_id":"2309.17234","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["s-abdelnabi/llm-deliberation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"split-and-merge-aligning-position-biases-in","title":"Split and Merge: Aligning Position Biases in LLM-based Evaluators","date":"2023-09-29","arxiv_id":"2310.01432","n_code_links":0,"syntology":null},{"paper":null,"slug":"ae-gpt-using-large-language-models-to-extract","title":"AE-GPT: Using Large Language Models to Extract Adverse Events from Surveillance Reports-A Use Case with Influenza Vaccine Adverse Events","date":"2023-09-28","arxiv_id":"2309.16150","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-fathom-benchmarking-large-language-models","slug":"gpt-fathom-benchmarking-large-language-models","title":"GPT-Fathom: Benchmarking Large Language Models to Decipher the Evolutionary Path towards GPT-4 and Beyond","date":"2023-09-28","arxiv_id":"2309.16583","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gpt-fathom/gpt-fathom"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"stress-testing-chain-of-thought-prompting-for","title":"Stress Testing Chain-of-Thought Prompting for Large Language Models","date":"2023-09-28","arxiv_id":"2309.16621","n_code_links":0,"syntology":null},{"paper":"/paper/nlpbench-evaluating-large-language-models-on","slug":"nlpbench-evaluating-large-language-models-on","title":"NLPBench: Evaluating Large Language Models on Solving NLP Problems","date":"2023-09-27","arxiv_id":"2309.15630","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-small-language-models-with-prompt","title":"Exploring Small Language Models with Prompt-Learning Paradigm for Efficient Domain-Specific Text Classification","date":"2023-09-26","arxiv_id":"2309.14779","n_code_links":0,"syntology":null},{"paper":"/paper/how-to-catch-an-ai-liar-lie-detection-in","slug":"how-to-catch-an-ai-liar-lie-detection-in","title":"How to Catch an AI Liar: Lie Detection in Black-Box LLMs by Asking Unrelated Questions","date":"2023-09-26","arxiv_id":"2309.15840","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lorypack/llm-liedetector"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rankvicuna-zero-shot-listwise-document","slug":"rankvicuna-zero-shot-listwise-document","title":"RankVicuna: Zero-Shot Listwise Document Reranking with Open-Source Large Language Models","date":"2023-09-26","arxiv_id":"2309.15088","n_code_links":3,"syntology":null},{"paper":"/paper/supersonic-learning-to-generate-source-code","slug":"supersonic-learning-to-generate-source-code","title":"Supersonic: Learning to Generate Source Code Optimizations in C/C++","date":"2023-09-26","arxiv_id":"2309.14846","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-cognitive-maps-and-planning-in","title":"Evaluating Cognitive Maps and Planning in Large Language Models with CogEval","date":"2023-09-25","arxiv_id":"2309.15129","n_code_links":0,"syntology":null},{"paper":null,"slug":"watch-your-language-large-language-models-and","title":"Watch Your Language: Investigating Content Moderation with Large Language Models","date":"2023-09-25","arxiv_id":"2309.14517","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-the-most-sinfully-decadent-cake-ever","title":"Does the \"most sinfully decadent cake ever\" taste good? Answering Yes/No Questions from Figurative Contexts","date":"2023-09-24","arxiv_id":"2309.13748","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-chat-about-boring-problems-studying-gpt","title":"A Chat About Boring Problems: Studying GPT-based text normalization","date":"2023-09-23","arxiv_id":"2309.13426","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-cognitive","title":"Probing the Moral Development of Large Language Models through Defining Issues Test","date":"2023-09-23","arxiv_id":"2309.13356","n_code_links":0,"syntology":null},{"paper":null,"slug":"benllmeval-a-comprehensive-evaluation-into","title":"BenLLMEval: A Comprehensive Evaluation into the Potentials and Pitfalls of Large Language Models on Bengali NLP","date":"2023-09-22","arxiv_id":"2309.13173","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-emotion-estimation-from-image","title":"Contextual Emotion Estimation from Image Captions","date":"2023-09-22","arxiv_id":"2309.13136","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-are-also-good","title":"Large Language Models Are Also Good Prototypical Commonsense Reasoners","date":"2023-09-22","arxiv_id":"2309.13165","n_code_links":0,"syntology":null},{"paper":"/paper/a-chinese-prompt-attack-dataset-for-llms-with","slug":"a-chinese-prompt-attack-dataset-for-llms-with","title":"Goal-Oriented Prompt Attack and Safety Evaluation for LLMs","date":"2023-09-21","arxiv_id":"2309.11830","n_code_links":2,"syntology":null},{"paper":"/paper/metamath-bootstrap-your-own-mathematical","slug":"metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","arxiv_id":"2309.12284","n_code_links":1,"syntology":{"ran":15,"of":22,"n_ran_checked":1,"n_instrument":14,"unverified":7,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 14 where Syntology's instrument failed) · 7 unverified","official":{"repos":["meta-math/MetaMath"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/tart-a-plug-and-play-transformer-module-for","slug":"tart-a-plug-and-play-transformer-module-for","title":"TART: A plug-and-play Transformer module for task-agnostic reasoning","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/the-cambridge-law-corpus-a-corpus-for-legal-1","slug":"the-cambridge-law-corpus-a-corpus-for-legal-1","title":"The Cambridge Law Corpus: A Dataset for Legal AI Research","date":"2023-09-21","arxiv_id":"2309.12269","n_code_links":0,"syntology":null},{"paper":"/paper/the-reversal-curse-llms-trained-on-a-is-b","slug":"the-reversal-curse-llms-trained-on-a-is-b","title":"The Reversal Curse: LLMs trained on \"A is B\" fail to learn \"B is A\"","date":"2023-09-21","arxiv_id":"2309.12288","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lukasberglund/reversal_curse"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"toa-task-oriented-active-vqa","title":"TOA: Task-oriented Active VQA","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-paradigm-shift-in-machine-translation","slug":"a-paradigm-shift-in-machine-translation","title":"A Paradigm Shift in Machine Translation: Boosting Translation Performance of Large Language Models","date":"2023-09-20","arxiv_id":"2309.11674","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fe1ixxu/alma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/controlled-generation-with-prompt-insertion","slug":"controlled-generation-with-prompt-insertion","title":"Controlled Generation with Prompt Insertion for Natural Language Explanations in Grammatical Error Correction","date":"2023-09-20","arxiv_id":"2309.11439","n_code_links":1,"syntology":null},{"paper":"/paper/design-of-chain-of-thought-in-math-problem","slug":"design-of-chain-of-thought-in-math-problem","title":"Design of Chain-of-Thought in Math Problem Solving","date":"2023-09-20","arxiv_id":"2309.11054","n_code_links":1,"syntology":null},{"paper":null,"slug":"fictional-worlds-real-connections-developing","title":"Fictional Worlds, Real Connections: Developing Community Storytelling Social Chatbots through LLMs","date":"2023-09-20","arxiv_id":"2309.11478","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-in-mafia-like-game-simulation","title":"Generative AI in Mafia-like Game Simulation","date":"2023-09-20","arxiv_id":"2309.11672","n_code_links":0,"syntology":null},{"paper":"/paper/safurai-001-new-qualitative-approach-for-code","slug":"safurai-001-new-qualitative-approach-for-code","title":"Safurai 001: New Qualitative Approach for Code LLM Evaluation","date":"2023-09-20","arxiv_id":"2309.11385","n_code_links":1,"syntology":null},{"paper":null,"slug":"language-as-the-medium-multimodal-video","title":"Language as the Medium: Multimodal Video Classification through text only","date":"2023-09-19","arxiv_id":"2309.10783","n_code_links":0,"syntology":null},{"paper":null,"slug":"writer-defined-ai-personas-for-on-demand","title":"Writer-Defined AI Personas for On-Demand Feedback Generation","date":"2023-09-19","arxiv_id":"2309.10433","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-gpt-3-for-anti-cancer-drug","title":"Evaluation of GPT-3 for Anti-Cancer Drug Sensitivity Prediction","date":"2023-09-18","arxiv_id":"2309.10016","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-decoding-improves-reasoning-in","title":"Contrastive Decoding Improves Reasoning in Large Language Models","date":"2023-09-17","arxiv_id":"2309.09117","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-large-gpt-models-discover-moral-dimensions","title":"Do Large GPT Models Discover Moral Dimensions in Language Representations? A Topological Study Of Sentence Embeddings","date":"2023-09-17","arxiv_id":"2309.09397","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-cooking-recipes-to-robot-task-trees","title":"From Cooking Recipes to Robot Task Trees -- Improving Planning Correctness and Task Efficiency by Leveraging LLMs with a Knowledge Network","date":"2023-09-17","arxiv_id":"2309.09181","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoder-only-architecture-for-speech","title":"Decoder-only Architecture for Speech Recognition with CTC Prompts and Text Data Augmentation","date":"2023-09-16","arxiv_id":"2309.08876","n_code_links":0,"syntology":null},{"paper":"/paper/struc-bench-are-large-language-models-really","slug":"struc-bench-are-large-language-models-really","title":"Struc-Bench: Are Large Language Models Really Good at Generating Complex Structured Data?","date":"2023-09-16","arxiv_id":"2309.08963","n_code_links":1,"syntology":null},{"paper":"/paper/advancing-the-evaluation-of-traditional","slug":"advancing-the-evaluation-of-traditional","title":"Advancing the Evaluation of Traditional Chinese Language Models: Towards a Comprehensive Benchmark Suite","date":"2023-09-15","arxiv_id":"2309.08448","n_code_links":1,"syntology":null},{"paper":"/paper/casteist-but-not-racist-quantifying","slug":"casteist-but-not-racist-quantifying","title":"Indian-BhED: A Dataset for Measuring India-Centric Biases in Large Language Models","date":"2023-09-15","arxiv_id":"2309.08573","n_code_links":1,"syntology":null},{"paper":"/paper/connecting-large-language-models-with","slug":"connecting-large-language-models-with","title":"EvoPrompt: Connecting LLMs with Evolutionary Algorithms Yields Powerful Prompt Optimizers","date":"2023-09-15","arxiv_id":"2309.08532","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":3,"n_instrument":8,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/iclef-in-context-learning-with-expert","slug":"iclef-in-context-learning-with-expert","title":"ICLEF: In-Context Learning with Expert Feedback for Explainable Style Transfer","date":"2023-09-15","arxiv_id":"2309.08583","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-for-failure-mode","slug":"large-language-models-for-failure-mode","title":"Large Language Models for Failure Mode Classification: An Investigation","date":"2023-09-15","arxiv_id":"2309.08181","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-empirical-evaluation-of-prompting","title":"An Empirical Evaluation of Prompting Strategies for Large Language Models in Zero-Shot Clinical Natural Language Processing","date":"2023-09-14","arxiv_id":"2309.08008","n_code_links":0,"syntology":null},{"paper":null,"slug":"two-timin-repairing-smart-contracts-with-a","title":"Two Timin': Repairing Smart Contracts With A Two-Layered Approach","date":"2023-09-14","arxiv_id":"2309.07841","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-can-infer-psychological","title":"Large Language Models Can Infer Psychological Dispositions of Social Media Users","date":"2023-09-13","arxiv_id":"2309.08631","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-llama-2-and-gpt-3-llms-for-hpc","title":"Comparing Llama-2 and GPT-3 LLMs for HPC kernels generation","date":"2023-09-12","arxiv_id":"2309.07103","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-large-language-models-for-ontology","slug":"exploring-large-language-models-for-ontology","title":"Exploring Large Language Models for Ontology Alignment","date":"2023-09-12","arxiv_id":"2309.07172","n_code_links":1,"syntology":null}],"record_sha256":"6f20ba62b6a4f088a962bc7e0f8d34b77c281e00eb40764f3b0f25e37e2df2f7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}