{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/10","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":20,"rows_per_page":100,"rows":[901,1000],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/9","next":"/method/gpt-3/papers/11","papers":[{"paper":"/paper/evil-geniuses-delving-into-the-safety-of-llm","slug":"evil-geniuses-delving-into-the-safety-of-llm","title":"Evil Geniuses: Delving into the Safety of LLM-based Agents","date":"2023-11-20","arxiv_id":"2311.11855","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["t1ans1r/evil-geniuses"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-to-use-large-language-models-for-text","slug":"how-to-use-large-language-models-for-text","title":"Towards Human-Level Text Coding with LLMs: The Case of Fatherhood Roles in Public Policy Documents","date":"2023-11-20","arxiv_id":"2311.11844","n_code_links":1,"syntology":null},{"paper":null,"slug":"refactoring-programs-using-large-language","title":"Refactoring Programs Using Large Language Models with Few-Shot Examples","date":"2023-11-20","arxiv_id":"2311.11690","n_code_links":0,"syntology":null},{"paper":null,"slug":"spot-the-bot-distinguishing-human-written-and","title":"Spot the Bot: Distinguishing Human-Written and Bot-Generated Texts Using Clustering and Information Theory Techniques","date":"2023-11-19","arxiv_id":"2311.11441","n_code_links":0,"syntology":null},{"paper":null,"slug":"behavior-optimized-image-generation","title":"Behavior Optimized Image Generation","date":"2023-11-18","arxiv_id":"2311.10995","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancements-in-generative-ai-a-comprehensive","title":"Advancements in Generative AI: A Comprehensive Review of GANs, GPT, Autoencoders, Diffusion Model, and Transformers","date":"2023-11-17","arxiv_id":"2311.10242","n_code_links":0,"syntology":null},{"paper":"/paper/camels-in-a-changing-climate-enhancing-lm","slug":"camels-in-a-changing-climate-enhancing-lm","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","date":"2023-11-17","arxiv_id":"2311.10702","n_code_links":3,"syntology":null},{"paper":null,"slug":"fumbling-in-babel-an-investigation-into","title":"Fumbling in Babel: An Investigation into ChatGPT's Language Identification Ability","date":"2023-11-16","arxiv_id":"2311.09696","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-for-hate-speech-detection","title":"Generative AI for Hate Speech Detection: Evaluation and Findings","date":"2023-11-16","arxiv_id":"2311.09993","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-still-wins-over-llm-an-empirical-study","title":"Human Still Wins over LLM: An Empirical Study of Active Learning on Domain-Specific Annotation Tasks","date":"2023-11-16","arxiv_id":"2311.09825","n_code_links":0,"syntology":null},{"paper":"/paper/intervenor-prompt-the-coding-ability-of-large","slug":"intervenor-prompt-the-coding-ability-of-large","title":"INTERVENOR: Prompting the Coding Ability of Large Language Models with the Interactive Chain of Repair","date":"2023-11-16","arxiv_id":"2311.09868","n_code_links":1,"syntology":null},{"paper":"/paper/knowledgemath-knowledge-intensive-math-word","slug":"knowledgemath-knowledge-intensive-math-word","title":"FinanceMath: Knowledge-Intensive Math Reasoning in Finance Domains","date":"2023-11-16","arxiv_id":"2311.09797","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yale-nlp/knowledgemath"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-retrieval-augmentation-and-the-limitations","title":"On Retrieval Augmentation and the Limitations of Language Model Training","date":"2023-11-16","arxiv_id":"2311.09615","n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-privacy-risks-in-online-self","title":"Reducing Privacy Risks in Online Self-Disclosures with Language Models","date":"2023-11-16","arxiv_id":"2311.09538","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-follow-concept","slug":"can-large-language-models-follow-concept","title":"Can Large Language Models Follow Concept Annotation Guidelines? A Case Study on Scientific and Financial Domains","date":"2023-11-15","arxiv_id":"2311.08704","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-gender-bias-in-the-translation-of","title":"Evaluating Gender Bias in the Translation of Gender-Neutral Languages into English","date":"2023-11-15","arxiv_id":"2311.08836","n_code_links":0,"syntology":null},{"paper":"/paper/tooltalk-evaluating-tool-usage-in-a","slug":"tooltalk-evaluating-tool-usage-in-a","title":"ToolTalk: Evaluating Tool-Usage in a Conversational Setting","date":"2023-11-15","arxiv_id":"2311.10775","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/we-demand-justice-towards-grounding-political","slug":"we-demand-justice-towards-grounding-political","title":"\"We Demand Justice!\": Towards Social Context Grounding of Political Texts","date":"2023-11-15","arxiv_id":"2311.09106","n_code_links":1,"syntology":null},{"paper":null,"slug":"cpopqa-ranking-cultural-concept-popularity-by","title":"CPopQA: Ranking Cultural Concept Popularity by LLMs","date":"2023-11-14","arxiv_id":"2311.07897","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-llms-on-document-based-qa-exact","title":"Evaluating LLMs on Document-Based QA: Exact Answer Selection and Numerical Extraction using Cogtale dataset","date":"2023-11-14","arxiv_id":"2311.07878","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-are-better-bug-detector","slug":"language-models-are-better-bug-detector","title":"Language Models are Better Bug Detector Through Code-Pair Classification","date":"2023-11-14","arxiv_id":"2311.07957","n_code_links":1,"syntology":null},{"paper":"/paper/do-large-language-models-and-humans-have","slug":"do-large-language-models-and-humans-have","title":"Do large language models and humans have similar behaviors in causal inference with script knowledge?","date":"2023-11-13","arxiv_id":"2311.07311","n_code_links":1,"syntology":null},{"paper":"/paper/it-s-not-easy-being-wrong-evaluating-process","slug":"it-s-not-easy-being-wrong-evaluating-process","title":"It's Not Easy Being Wrong: Large Language Models Struggle with Process of Elimination Reasoning","date":"2023-11-13","arxiv_id":"2311.07532","n_code_links":1,"syntology":null},{"paper":null,"slug":"megaverse-benchmarking-large-language-models","title":"MEGAVERSE: Benchmarking Large Language Models Across Languages, Modalities, Models and Tasks","date":"2023-11-13","arxiv_id":"2311.07463","n_code_links":0,"syntology":null},{"paper":null,"slug":"speech-based-slot-filling-using-large","title":"Speech-based Slot Filling using Large Language Models","date":"2023-11-13","arxiv_id":"2311.07418","n_code_links":0,"syntology":null},{"paper":"/paper/steer-unified-style-transfer-with-expert","slug":"steer-unified-style-transfer-with-expert","title":"STEER: Unified Style Transfer with Expert Reinforcement","date":"2023-11-13","arxiv_id":"2311.07167","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-complex-to-simple-unraveling-the","title":"From Complex to Simple: Unraveling the Cognitive Tree for Reasoning with Small Language Models","date":"2023-11-12","arxiv_id":"2311.06754","n_code_links":0,"syntology":null},{"paper":null,"slug":"giellm-japanese-general-information","title":"GIELLM: Japanese General Information Extraction Large Language Model Utilizing Mutual Reinforcement Effect","date":"2023-11-12","arxiv_id":"2311.06838","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-explain-teaching-large-language-models","title":"Large Language Models are In-context Teachers for Knowledge Reasoning","date":"2023-11-12","arxiv_id":"2311.06985","n_code_links":0,"syntology":null},{"paper":"/paper/data-contamination-quiz-a-tool-to-detect-and","slug":"data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shahriargolchin/dcq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-and-prompt-engineering","title":"Large Language Models and Prompt Engineering for Biomedical Query Focused Multi-Document Summarisation","date":"2023-11-09","arxiv_id":"2311.05169","n_code_links":0,"syntology":null},{"paper":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-large-language-models-in","title":"Evaluating Large Language Models in Ophthalmology","date":"2023-11-07","arxiv_id":"2311.04933","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-and-mitigating-vulnerabilities-in","title":"Identifying and Mitigating Vulnerabilities in LLM-Integrated Applications","date":"2023-11-07","arxiv_id":"2311.16153","n_code_links":0,"syntology":null},{"paper":"/paper/deepinception-hypnotize-large-language-model","slug":"deepinception-hypnotize-large-language-model","title":"DeepInception: Hypnotize Large Language Model to Be Jailbreaker","date":"2023-11-06","arxiv_id":"2311.03191","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tmlr-group/deepinception"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"in-context-learning-for-knowledge-base","title":"In-Context Learning for Knowledge Base Question Answering for Unmanned Systems based on Large Language Models","date":"2023-11-06","arxiv_id":"2311.02956","n_code_links":0,"syntology":null},{"paper":"/paper/unraveling-downstream-gender-bias-from-large","slug":"unraveling-downstream-gender-bias-from-large","title":"Unraveling Downstream Gender Bias from Large Language Models: A Study on AI Educational Writing Assistance","date":"2023-11-06","arxiv_id":"2311.03311","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-potential-of-leading-large","title":"Evaluating the Potential of Leading Large Language Models in Reasoning Biology Questions","date":"2023-11-05","arxiv_id":"2311.07582","n_code_links":0,"syntology":null},{"paper":"/paper/extraction-of-atypical-aspects-from-customer","slug":"extraction-of-atypical-aspects-from-customer","title":"Extraction of Atypical Aspects from Customer Reviews: Datasets and Experiments with Language Models","date":"2023-11-05","arxiv_id":"2311.02702","n_code_links":2,"syntology":null},{"paper":null,"slug":"uid-as-a-guiding-metric-for-automated","title":"UID as a Guiding Metric for Automated Authorship Obfuscation","date":"2023-11-05","arxiv_id":"2312.03709","n_code_links":0,"syntology":null},{"paper":"/paper/automating-governing-knowledge-commons-and","slug":"automating-governing-knowledge-commons-and","title":"Automating Governing Knowledge Commons and Contextual Integrity (GKC-CI) Privacy Policy Annotations with Large Language Models","date":"2023-11-03","arxiv_id":"2311.02192","n_code_links":1,"syntology":null},{"paper":null,"slug":"cosmic-data-efficient-instruction-tuning-for","title":"COSMIC: Data Efficient Instruction-tuning For Speech In-Context Learning","date":"2023-11-03","arxiv_id":"2311.02248","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-black-box-adversarial-attacks-on","slug":"efficient-black-box-adversarial-attacks-on","title":"Efficient Black-Box Adversarial Attacks on Neural Text Detectors","date":"2023-11-03","arxiv_id":"2311.01873","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-numerical-reasoning","title":"Exploring the Numerical Reasoning Capabilities of Language Models: A Comprehensive Analysis on Tabular Data","date":"2023-11-03","arxiv_id":"2311.02216","n_code_links":0,"syntology":null},{"paper":"/paper/long-story-short-a-summarize-then-search","slug":"long-story-short-a-summarize-then-search","title":"Long Story Short: a Summarize-then-Search Method for Long Video Question Answering","date":"2023-11-02","arxiv_id":"2311.01233","n_code_links":1,"syntology":null},{"paper":null,"slug":"measuring-five-accountable-talk-moves-to","title":"Measuring Five Accountable Talk Moves to Improve Instruction at Scale","date":"2023-11-02","arxiv_id":"2311.10749","n_code_links":0,"syntology":null},{"paper":null,"slug":"server-side-rescoring-of-spoken-entity","title":"Server-side Rescoring of Spoken Entity-centric Knowledge Queries for Virtual Assistants","date":"2023-11-02","arxiv_id":"2311.01398","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-reliable-judges-a","title":"Are Large Language Models Reliable Judges? A Study on the Factuality Evaluation Capabilities of LLMs","date":"2023-11-01","arxiv_id":"2311.00681","n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-training-and-fine-tuning-for","title":"Continuous Training and Fine-tuning for Domain-Specific Language Models in Medical Question Answering","date":"2023-11-01","arxiv_id":"2311.00204","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-gpt-powerful-enough-to-analyze-the","title":"Is GPT Powerful Enough to Analyze the Emotions of Memes?","date":"2023-11-01","arxiv_id":"2311.00223","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-lexical-simplification-with","slug":"unsupervised-lexical-simplification-with","title":"Unsupervised Lexical Simplification with Context Augmentation","date":"2023-11-01","arxiv_id":"2311.00310","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-large-language-models-solve-verbal","title":"Do large language models solve verbal analogies like children do?","date":"2023-10-31","arxiv_id":"2310.20384","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-gpt-4-pass-the-turing-test","title":"Does GPT-4 pass the Turing test?","date":"2023-10-31","arxiv_id":"2310.20216","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-classification-of-student-help","title":"Efficient Classification of Student Help Requests in Programming Courses Using Large Language Models","date":"2023-10-31","arxiv_id":"2310.20105","n_code_links":0,"syntology":null},{"paper":null,"slug":"interactive-multi-fidelity-learning-for-cost","title":"Interactive Multi-fidelity Learning for Cost-effective Adaptation of Language Model with Sparse Human Supervision","date":"2023-10-31","arxiv_id":"2310.20153","n_code_links":0,"syntology":null},{"paper":"/paper/psycot-psychological-questionnaire-as","slug":"psycot-psychological-questionnaire-as","title":"PsyCoT: Psychological Questionnaire as Powerful Chain-of-Thought for Personality Detection","date":"2023-10-31","arxiv_id":"2310.20256","n_code_links":1,"syntology":null},{"paper":"/paper/interpretable-by-design-text-classification","slug":"interpretable-by-design-text-classification","title":"Interpretable-by-Design Text Understanding with Iteratively Generated Concept Bottleneck","date":"2023-10-30","arxiv_id":"2310.19660","n_code_links":1,"syntology":null},{"paper":null,"slug":"remember-what-you-did-so-you-know-what-to-do","title":"Remember what you did so you know what to do next","date":"2023-10-30","arxiv_id":"2311.01468","n_code_links":0,"syntology":null},{"paper":"/paper/eticor-corpus-for-analyzing-llms-for","slug":"eticor-corpus-for-analyzing-llms-for","title":"EtiCor: Corpus for Analyzing LLMs for Etiquettes","date":"2023-10-29","arxiv_id":"2310.18974","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-for-aspect-based","slug":"large-language-models-for-aspect-based","title":"Large language models for aspect-based sentiment analysis","date":"2023-10-27","arxiv_id":"2310.18025","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qagentur/absa_llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/offmix-3l-a-novel-code-mixed-dataset-in","slug":"offmix-3l-a-novel-code-mixed-dataset-in","title":"OffMix-3L: A Novel Code-Mixed Dataset in Bangla-English-Hindi for Offensive Language Identification","date":"2023-10-27","arxiv_id":"2310.18387","n_code_links":1,"syntology":null},{"paper":"/paper/sentmix-3l-a-bangla-english-hindi-code-mixed","slug":"sentmix-3l-a-bangla-english-hindi-code-mixed","title":"SentMix-3L: A Bangla-English-Hindi Code-Mixed Dataset for Sentiment Analysis","date":"2023-10-27","arxiv_id":"2310.18023","n_code_links":1,"syntology":null},{"paper":null,"slug":"fedpeat-convergence-of-federated-learning","title":"FedPEAT: Convergence of Federated Learning, Parameter-Efficient Fine Tuning, and Emulator Assisted Tuning for Artificial Intelligence Foundation Models with Mobile Edge Computing","date":"2023-10-26","arxiv_id":"2310.17491","n_code_links":0,"syntology":null},{"paper":"/paper/in-context-learning-dynamics-with-random","slug":"in-context-learning-dynamics-with-random","title":"In-Context Learning Dynamics with Random Binary Sequences","date":"2023-10-26","arxiv_id":"2310.17639","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ebigelow/icl-random-binary"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"you-are-an-expert-linguistic-annotator-limits","title":"\"You Are An Expert Linguistic Annotator\": Limits of LLMs as Analyzers of Abstract Meaning Representation","date":"2023-10-26","arxiv_id":"2310.17793","n_code_links":0,"syntology":null},{"paper":null,"slug":"boost-harnessing-black-box-control-to-boost","title":"BOOST: Harnessing Black-Box Control to Boost Commonsense in LMs' Generation","date":"2023-10-25","arxiv_id":"2310.17054","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-stumpers-large-language-models-vs","title":"Decoding Stumpers: Large Language Models vs. Human Problem-Solvers","date":"2023-10-25","arxiv_id":"2310.16411","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-well-can-machine-generated-texts-be","title":"How well can machine-generated texts be identified and can language models be trained to avoid identification?","date":"2023-10-25","arxiv_id":"2310.16992","n_code_links":0,"syntology":null},{"paper":null,"slug":"muslim-violence-bias-persists-in-debiased-gpt","title":"Muslim-Violence Bias Persists in Debiased GPT Models","date":"2023-10-25","arxiv_id":"2310.18368","n_code_links":0,"syntology":null},{"paper":null,"slug":"r-3-prompting-review-rephrase-and-resolve-for","title":"R$^3$ Prompting: Review, Rephrase and Resolve for Chain-of-Thought Reasoning in Large Language Models under Noisy Context","date":"2023-10-25","arxiv_id":"2310.16535","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-communication-theory-perspective-on","title":"A Communication Theory Perspective on Prompting Engineering Methods for Large Language Models","date":"2023-10-24","arxiv_id":"2310.18358","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-enhanced-auto-correction-of-programming","title":"AI-enhanced Auto-correction of Programming Exercises: How Effective is GPT-3.5?","date":"2023-10-24","arxiv_id":"2311.10737","n_code_links":0,"syntology":null},{"paper":"/paper/background-summarization-of-event-timelines","slug":"background-summarization-of-event-timelines","title":"Background Summarization of Event Timelines","date":"2023-10-24","arxiv_id":"2310.16197","n_code_links":1,"syntology":null},{"paper":null,"slug":"dissecting-in-context-learning-of","title":"Dissecting In-Context Learning of Translations in GPTs","date":"2023-10-24","arxiv_id":"2310.15987","n_code_links":0,"syntology":null},{"paper":"/paper/fighting-fire-with-fire-the-dual-role-of-llms","slug":"fighting-fire-with-fire-the-dual-role-of-llms","title":"Fighting Fire with Fire: The Dual Role of LLMs in Crafting and Detecting Elusive Disinformation","date":"2023-10-24","arxiv_id":"2310.15515","n_code_links":1,"syntology":null},{"paper":"/paper/the-janus-interface-how-fine-tuning-in-large","slug":"the-janus-interface-how-fine-tuning-in-large","title":"The Janus Interface: How Fine-Tuning in Large Language Models Amplifies the Privacy Risks","date":"2023-10-24","arxiv_id":"2310.15469","n_code_links":1,"syntology":null},{"paper":null,"slug":"what-makes-it-ok-to-set-a-fire-iterative-self","title":"What Makes it Ok to Set a Fire? Iterative Self-distillation of Contexts and Rationales for Disambiguating Defeasible Social and Moral Situations","date":"2023-10-24","arxiv_id":"2310.15431","n_code_links":0,"syntology":null},{"paper":null,"slug":"causal-inference-using-llm-guided-discovery","title":"Causal Inference Using LLM-Guided Discovery","date":"2023-10-23","arxiv_id":"2310.15117","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-spatial-understanding-of-large","slug":"evaluating-spatial-understanding-of-large","title":"Evaluating Spatial Understanding of Large Language Models","date":"2023-10-23","arxiv_id":"2310.14540","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["runopti/spatialevalllm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-knowledge-base-completion","title":"Evaluating the Knowledge Base Completion Potential of GPT","date":"2023-10-23","arxiv_id":"2310.14771","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-as-an-effective-zero-shot-evaluator-for","title":"GPT-4 as an Effective Zero-Shot Evaluator for Scientific Figure Captions","date":"2023-10-23","arxiv_id":"2310.15405","n_code_links":0,"syntology":null},{"paper":null,"slug":"instructexcel-a-benchmark-for-natural","title":"InstructExcel: A Benchmark for Natural Language Instruction in Excel","date":"2023-10-23","arxiv_id":"2310.14495","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-hallucinate-but-may-excel-at","slug":"language-models-hallucinate-but-may-excel-at","title":"Language Models Hallucinate, but May Excel at Fact Verification","date":"2023-10-23","arxiv_id":"2310.14564","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jianguanthu/llmforfv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/linc-a-neurosymbolic-approach-for-logical","slug":"linc-a-neurosymbolic-approach-for-logical","title":"LINC: A Neurosymbolic Approach for Logical Reasoning by Combining Language Models with First-Order Logic Provers","date":"2023-10-23","arxiv_id":"2310.15164","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":0,"n_instrument":1,"unverified":6,"pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["benlipkin/linc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-in-the-loop-leveraging-large-language","slug":"llm-in-the-loop-leveraging-large-language","title":"LLM-in-the-loop: Leveraging Large Language Model for Thematic Analysis","date":"2023-10-23","arxiv_id":"2310.15100","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sjdai/llm-thematic-analysis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/teleqna-a-benchmark-dataset-to-assess-large","slug":"teleqna-a-benchmark-dataset-to-assess-large","title":"TeleQnA: A Benchmark Dataset to Assess Large Language Models Telecommunications Knowledge","date":"2023-10-23","arxiv_id":"2310.15051","n_code_links":1,"syntology":null},{"paper":"/paper/can-language-models-laugh-at-youtube-short","slug":"can-language-models-laugh-at-youtube-short","title":"Can Language Models Laugh at YouTube Short-form Videos?","date":"2023-10-22","arxiv_id":"2310.14159","n_code_links":1,"syntology":null},{"paper":"/paper/is-chatgpt-a-game-changer-for-geocoding-a","slug":"is-chatgpt-a-game-changer-for-geocoding-a","title":"Is ChatGPT a game changer for geocoding -- a benchmark for geocoding address parsing techniques","date":"2023-10-22","arxiv_id":"2310.14360","n_code_links":1,"syntology":null},{"paper":"/paper/text-generation-for-dataset-augmentation-in","slug":"text-generation-for-dataset-augmentation-in","title":"Text generation for dataset augmentation in security classification tasks","date":"2023-10-22","arxiv_id":"2310.14429","n_code_links":1,"syntology":null},{"paper":null,"slug":"haterephrase-zero-and-few-shot-reduction-of","title":"HateRephrase: Zero- and Few-Shot Reduction of Hate Intensity in Online Posts using Large Language Models","date":"2023-10-21","arxiv_id":"2310.13985","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-baseline-for-knowledge-based-visual","slug":"a-simple-baseline-for-knowledge-based-visual","title":"A Simple Baseline for Knowledge-Based Visual Question Answering","date":"2023-10-20","arxiv_id":"2310.13570","n_code_links":0,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/cache-me-if-you-can-an-online-cost-aware","slug":"cache-me-if-you-can-an-online-cost-aware","title":"Cache me if you Can: an Online Cost-aware Teacher-Student framework to Reduce the Calls to Large Language Models","date":"2023-10-20","arxiv_id":"2310.13395","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stoyian/OCaTS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"design-inclusive-language-models-for","title":"She had Cobalt Blue Eyes: Prompt Testing to Create Aligned and Sustainable Language Models","date":"2023-10-20","arxiv_id":"2310.18333","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-perils-promises-of-fact-checking-with","title":"The Perils & Promises of Fact-checking with Large Language Models","date":"2023-10-20","arxiv_id":"2310.13549","n_code_links":0,"syntology":null},{"paper":null,"slug":"wordart-designer-user-driven-artistic","title":"WordArt Designer: User-Driven Artistic Typography Synthesis using Large Language Models","date":"2023-10-20","arxiv_id":"2310.18332","n_code_links":0,"syntology":null},{"paper":"/paper/agenttuning-enabling-generalized-agent","slug":"agenttuning-enabling-generalized-agent","title":"AgentTuning: Enabling Generalized Agent Abilities for LLMs","date":"2023-10-19","arxiv_id":"2310.12823","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thudm/agenttuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-exploration-of-in-context-learning-for","title":"Exploring In-Context Learning of Textless Speech Language Model for Speech Classification Tasks","date":"2023-10-19","arxiv_id":"2310.12477","n_code_links":0,"syntology":null},{"paper":null,"slug":"experimental-narratives-a-comparison-of-human","title":"Experimental Narratives: A Comparison of Human Crowdsourced Storytelling and AI Storytelling","date":"2023-10-19","arxiv_id":"2310.12902","n_code_links":0,"syntology":null},{"paper":"/paper/product-attribute-value-extraction-using","slug":"product-attribute-value-extraction-using","title":"ExtractGPT: Exploring the Potential of Large Language Models for Product Attribute Value Extraction","date":"2023-10-19","arxiv_id":"2310.12537","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-symbol-binding-ability-of","title":"Evaluating the Symbol Binding Ability of Large Language Models for Multiple-Choice Questions in Vietnamese General Education","date":"2023-10-18","arxiv_id":"2310.12059","n_code_links":0,"syntology":null}],"record_sha256":"14290a259068b780abc473c0a61f8c1135f6b3f5685b3c447ef66fbedd6ec9da","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}