{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/27","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":27,"pages_in_order":29,"rows_per_page":100,"rows":[2601,2700],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/26","next":"/method/gpt-4/papers/28","papers":[{"paper":"/paper/unleashing-cognitive-synergy-in-large","slug":"unleashing-cognitive-synergy-in-large","title":"Unleashing the Emergent Cognitive Synergy in Large Language Models: A Task-Solving Agent through Multi-Persona Self-Collaboration","date":"2023-07-11","arxiv_id":"2307.05300","n_code_links":2,"syntology":null},{"paper":"/paper/chatgpt-for-digital-forensic-investigation","slug":"chatgpt-for-digital-forensic-investigation","title":"ChatGPT for Digital Forensic Investigation: The Good, The Bad, and The Unknown","date":"2023-07-10","arxiv_id":"2307.10195","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-the-efficacy-of-large-language","title":"Assessing the efficacy of large language models in generating accurate teacher responses","date":"2023-07-09","arxiv_id":"2307.04274","n_code_links":0,"syntology":null},{"paper":"/paper/svit-scaling-up-visual-instruction-tuning","slug":"svit-scaling-up-visual-instruction-tuning","title":"SVIT: Scaling up Visual Instruction Tuning","date":"2023-07-09","arxiv_id":"2307.04087","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-and-characterizing-large-language","title":"Exploring and Characterizing Large Language Models For Embedded System Development and Debugging","date":"2023-07-07","arxiv_id":"2307.03817","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-batteries-included","title":"Large Language Models as Batteries-Included Zero-Shot ESCO Skills Matchers","date":"2023-07-07","arxiv_id":"2307.03539","n_code_links":0,"syntology":null},{"paper":"/paper/teaching-arithmetic-to-small-transformers","slug":"teaching-arithmetic-to-small-transformers","title":"Teaching Arithmetic to Small Transformers","date":"2023-07-07","arxiv_id":"2307.03381","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lee-ny/teaching_arithmetic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/building-cooperative-embodied-agents","slug":"building-cooperative-embodied-agents","title":"Building Cooperative Embodied Agents Modularly with Large Language Models","date":"2023-07-05","arxiv_id":"2307.02485","n_code_links":2,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-gpt-4-and-human","title":"Comparative Analysis of GPT-4 and Human Graders in Evaluating Praise Given to Students in Synthetic Dialogues","date":"2023-07-05","arxiv_id":"2307.02018","n_code_links":0,"syntology":null},{"paper":"/paper/external-reasoning-towards-multi-large","slug":"external-reasoning-towards-multi-large","title":"External Reasoning: Towards Multi-Large-Language-Models Interchangeable Assistance with Human Feedback","date":"2023-07-05","arxiv_id":"2307.12057","n_code_links":1,"syntology":null},{"paper":"/paper/hoodwinked-deception-and-cooperation-in-a","slug":"hoodwinked-deception-and-cooperation-in-a","title":"Hoodwinked: Deception and Cooperation in a Text-Based Game for Language Models","date":"2023-07-05","arxiv_id":"2308.01404","n_code_links":1,"syntology":null},{"paper":"/paper/jailbroken-how-does-llm-safety-training-fail","slug":"jailbroken-how-does-llm-safety-training-fail","title":"Jailbroken: How Does LLM Safety Training Fail?","date":"2023-07-05","arxiv_id":"2307.02483","n_code_links":1,"syntology":null},{"paper":null,"slug":"open-source-large-language-models-outperform","title":"Open-Source LLMs for Text Annotation: A Practical Guide for Model Setting and Fine-Tuning","date":"2023-07-05","arxiv_id":"2307.02179","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-shutdown-avoidance-of-language","slug":"evaluating-shutdown-avoidance-of-language","title":"Evaluating Shutdown Avoidance of Language Models in Textual Scenarios","date":"2023-07-03","arxiv_id":"2307.00787","n_code_links":1,"syntology":null},{"paper":null,"slug":"harnessing-llms-in-curricular-design-using","title":"Harnessing LLMs in Curricular Design: Using GPT-4 to Support Authoring of Learning Objectives","date":"2023-06-30","arxiv_id":"2306.17459","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-are-effective-text","title":"Large Language Models are Effective Text Rankers with Pairwise Ranking Prompting","date":"2023-06-30","arxiv_id":"2306.17563","n_code_links":0,"syntology":null},{"paper":"/paper/preference-ranking-optimization-for-human","slug":"preference-ranking-optimization-for-human","title":"Preference Ranking Optimization for Human Alignment","date":"2023-06-30","arxiv_id":"2306.17492","n_code_links":1,"syntology":null},{"paper":"/paper/summqa-at-mediqa-chat-2023-in-context","slug":"summqa-at-mediqa-chat-2023-in-context","title":"SummQA at MEDIQA-Chat 2023:In-Context Learning with GPT-4 for Medical Summarization","date":"2023-06-30","arxiv_id":"2306.17384","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-negation-detection-assessment-of-gpts","title":"A negation detection assessment of GPTs: analysis with the xNot360 dataset","date":"2023-06-29","arxiv_id":"2306.16638","n_code_links":0,"syntology":null},{"paper":null,"slug":"cmath-can-your-language-model-pass-chinese","title":"CMATH: Can Your Language Model Pass Chinese Elementary School Math Test?","date":"2023-06-29","arxiv_id":"2306.16636","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-for-programming-education","title":"Generative AI for Programming Education: Benchmarking ChatGPT, GPT-4, and Human Tutors","date":"2023-06-29","arxiv_id":"2306.17156","n_code_links":0,"syntology":null},{"paper":"/paper/llavar-enhanced-visual-instruction-tuning-for","slug":"llavar-enhanced-visual-instruction-tuning-for","title":"LLaVAR: Enhanced Visual Instruction Tuning for Text-Rich Image Understanding","date":"2023-06-29","arxiv_id":"2306.17107","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["SALT-NLP/LLaVAR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/lyricwhiz-robust-multilingual-zero-shot","slug":"lyricwhiz-robust-multilingual-zero-shot","title":"LyricWhiz: Robust Multilingual Zero-shot Lyrics Transcription by Whispering to ChatGPT","date":"2023-06-29","arxiv_id":"2306.17103","n_code_links":1,"syntology":null},{"paper":"/paper/umass-bionlp-at-mediqa-chat-2023-can-llms","slug":"umass-bionlp-at-mediqa-chat-2023-can-llms","title":"UMASS_BioNLP at MEDIQA-Chat 2023: Can LLMs generate high-quality synthetic note-oriented doctor-patient conversations?","date":"2023-06-29","arxiv_id":"2306.16931","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["believewhat/dr.noteaid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automatic-calibration-and-error-correction","title":"Pareto Optimal Learning for Estimating Large Language Model Errors","date":"2023-06-28","arxiv_id":"2306.16564","n_code_links":0,"syntology":null},{"paper":"/paper/chatlaw-open-source-legal-large-language","slug":"chatlaw-open-source-legal-large-language","title":"Chatlaw: A Multi-Agent Collaborative Legal Assistant with Knowledge Graph Enhanced Mixture-of-Experts Large Language Model","date":"2023-06-28","arxiv_id":"2306.16092","n_code_links":1,"syntology":null},{"paper":"/paper/is-chatgpt-a-biomedical-expert-exploring-the","slug":"is-chatgpt-a-biomedical-expert-exploring-the","title":"Is ChatGPT a Biomedical Expert? -- Exploring the Zero-Shot Performance of Current GPT Models in Biomedical Tasks","date":"2023-06-28","arxiv_id":"2306.16108","n_code_links":1,"syntology":null},{"paper":null,"slug":"leveraging-gpt-4-for-food-effect","title":"Leveraging GPT-4 for Food Effect Summarization to Enhance Product-Specific Guidance Development via Iterative Prompting","date":"2023-06-28","arxiv_id":"2306.16275","n_code_links":0,"syntology":null},{"paper":"/paper/taqyim-evaluating-arabic-nlp-tasks-using","slug":"taqyim-evaluating-arabic-nlp-tasks-using","title":"Taqyim: Evaluating Arabic NLP Tasks Using ChatGPT Models","date":"2023-06-28","arxiv_id":"2306.16322","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-gpt-3-5-and-gpt-4-on-grammatical","title":"Evaluating GPT-3.5 and GPT-4 on Grammatical Error Correction for Brazilian Portuguese","date":"2023-06-27","arxiv_id":"2306.15788","n_code_links":0,"syntology":null},{"paper":"/paper/leandojo-theorem-proving-with-retrieval-1","slug":"leandojo-theorem-proving-with-retrieval-1","title":"LeanDojo: Theorem Proving with Retrieval-Augmented Language Models","date":"2023-06-27","arxiv_id":"2306.15626","n_code_links":3,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lean-dojo/leandojo","lean-dojo/leandojochatgpt","lean-dojo/reprover"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-multimodal-models-notes-on-cvpr-2023","title":"Large Multimodal Models: Notes on CVPR 2023 Tutorial","date":"2023-06-26","arxiv_id":"2306.14895","n_code_links":0,"syntology":null},{"paper":null,"slug":"lm4hpc-towards-effective-language-model","title":"LM4HPC: Towards Effective Language Model Application in High-Performance Computing","date":"2023-06-26","arxiv_id":"2306.14979","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-4-support-analysis-of-textual-data-in","title":"Can GPT-4 Support Analysis of Textual Data in Tasks Requiring Highly Specialized Domain Expertise?","date":"2023-06-24","arxiv_id":"2306.13906","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-express-their-uncertainty-an","slug":"can-llms-express-their-uncertainty-an","title":"Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs","date":"2023-06-22","arxiv_id":"2306.13063","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["miaoxiong2320/llm-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/cross-lingual-cross-temporal-summarization","slug":"cross-lingual-cross-temporal-summarization","title":"Cross-lingual Cross-temporal Summarization: Dataset, Models, Evaluation","date":"2023-06-22","arxiv_id":"2306.12916","n_code_links":1,"syntology":null},{"paper":"/paper/visual-adversarial-examples-jailbreak-large","slug":"visual-adversarial-examples-jailbreak-large","title":"Visual Adversarial Examples Jailbreak Aligned Large Language Models","date":"2023-06-22","arxiv_id":"2306.13213","n_code_links":1,"syntology":null},{"paper":"/paper/aries-a-corpus-of-scientific-paper-edits-made","slug":"aries-a-corpus-of-scientific-paper-edits-made","title":"ARIES: A Corpus of Scientific Paper Edits Made in Response to Peer Reviews","date":"2023-06-21","arxiv_id":"2306.12587","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/aries"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/deep-language-networks-joint-prompt-training","slug":"deep-language-networks-joint-prompt-training","title":"Joint Prompt Optimization of Stacked LLMs using Variational Inference","date":"2023-06-21","arxiv_id":"2306.12509","n_code_links":1,"syntology":{"ran":3,"of":19,"n_ran_checked":0,"n_instrument":3,"unverified":16,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 16 unverified","official":{"repos":["microsoft/deep-language-networks"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":16,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-based-models-meet-simulation-how-to","title":"GPT-Based Models Meet Simulation: How to Efficiently Use Large-Scale Pre-Trained Language Models Across Simulation Tasks","date":"2023-06-21","arxiv_id":"2306.13679","n_code_links":0,"syntology":null},{"paper":null,"slug":"decodingtrust-a-comprehensive-assessment-of","title":"DecodingTrust: A Comprehensive Assessment of Trustworthiness in GPT Models","date":"2023-06-20","arxiv_id":"2306.11698","n_code_links":0,"syntology":null},{"paper":null,"slug":"democratizing-llms-for-low-resource-languages","title":"Democratizing LLMs for Low-Resource Languages by Leveraging their English Dominant Abilities with Linguistically-Diverse Prompts","date":"2023-06-20","arxiv_id":"2306.11372","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-4-reticular-chemist-for-mof-discovery","slug":"gpt-4-reticular-chemist-for-mof-discovery","title":"A GPT-4 Reticular Chemist for Guiding MOF Discovery","date":"2023-06-20","arxiv_id":"2306.14915","n_code_links":1,"syntology":null},{"paper":null,"slug":"harnessing-the-power-of-adversarial-prompting","title":"Harnessing the Power of Adversarial Prompting and Large Language Models for Robust Hypothesis Generation in Astronomy","date":"2023-06-20","arxiv_id":"2306.11648","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-generate-better-than-your-llm","slug":"learning-to-generate-better-than-your-llm","title":"Learning to Generate Better Than Your LLM","date":"2023-06-20","arxiv_id":"2306.11816","n_code_links":1,"syntology":null},{"paper":null,"slug":"rm-prt-realistic-robotic-manipulation","title":"Surfer: Progressive Reasoning with World Models for Robotic Manipulation","date":"2023-06-20","arxiv_id":"2306.11335","n_code_links":0,"syntology":null},{"paper":"/paper/bayling-bridging-cross-lingual-alignment-and","slug":"bayling-bridging-cross-lingual-alignment-and","title":"BayLing: Bridging Cross-lingual Alignment and Instruction Following through Interactive Translation for Large Language Models","date":"2023-06-19","arxiv_id":"2306.10968","n_code_links":1,"syntology":null},{"paper":null,"slug":"temporal-data-meets-llm-explainable-financial","title":"Temporal Data Meets LLM -- Explainable Financial Time Series Forecasting","date":"2023-06-19","arxiv_id":"2306.11025","n_code_links":0,"syntology":null},{"paper":null,"slug":"news-verifiers-showdown-a-comparative","title":"News Verifiers Showdown: A Comparative Performance Evaluation of ChatGPT 3.5, ChatGPT 4.0, Bing AI, and Bard in News Fact-Checking","date":"2023-06-18","arxiv_id":"2306.17176","n_code_links":0,"syntology":null},{"paper":null,"slug":"snowman-a-million-scale-chinese-commonsense","title":"Snowman: A Million-scale Chinese Commonsense Knowledge Graph Distilled from Foundation Model","date":"2023-06-17","arxiv_id":"2306.10241","n_code_links":0,"syntology":null},{"paper":null,"slug":"ad-autogpt-an-autonomous-gpt-for-alzheimer-s","title":"AD-AutoGPT: An Autonomous GPT for Alzheimer's Disease Infodemiology","date":"2023-06-16","arxiv_id":"2306.10095","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-gpt-self-repair-for-code","slug":"demystifying-gpt-self-repair-for-code","title":"Is Self-Repair a Silver Bullet for Code Generation?","date":"2023-06-16","arxiv_id":"2306.09896","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-superhuman-models-with-consistency","slug":"evaluating-superhuman-models-with-consistency","title":"Evaluating Superhuman Models with Consistency Checks","date":"2023-06-16","arxiv_id":"2306.09983","n_code_links":2,"syntology":null},{"paper":null,"slug":"explaining-legal-concepts-with-augmented","title":"Explaining Legal Concepts with Augmented Large Language Models (GPT-4)","date":"2023-06-15","arxiv_id":"2306.09525","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-mit-mathematics-and-eecs","title":"Exploring the MIT Mathematics and EECS Curriculum Using Large Language Models","date":"2023-06-15","arxiv_id":"2306.08997","n_code_links":0,"syntology":null},{"paper":null,"slug":"thrilled-by-your-progress-large-language","title":"Thrilled by Your Progress! Large Language Models (GPT-4) No Longer Struggle to Pass Assessments in Higher Education Programming Courses","date":"2023-06-15","arxiv_id":"2306.10073","n_code_links":0,"syntology":null},{"paper":null,"slug":"tighter-prediction-intervals-for-causal","title":"Ensembled Prediction Intervals for Causal Outcomes Under Hidden Confounding","date":"2023-06-15","arxiv_id":"2306.09520","n_code_links":0,"syntology":null},{"paper":"/paper/arxiveri-automatic-table-verification-with","slug":"arxiveri-automatic-table-verification-with","title":"arXiVeri: Automatic table verification with GPT","date":"2023-06-13","arxiv_id":"2306.07968","n_code_links":1,"syntology":null},{"paper":"/paper/can-chatgpt-enable-its-the-case-of-mixed","slug":"can-chatgpt-enable-its-the-case-of-mixed","title":"Can ChatGPT Enable ITS? The Case of Mixed Traffic Control via Reinforcement Learning","date":"2023-06-13","arxiv_id":"2306.08094","n_code_links":1,"syntology":null},{"paper":"/paper/h2ogpt-democratizing-large-language-models","slug":"h2ogpt-democratizing-large-language-models","title":"h2oGPT: Democratizing Large Language Models","date":"2023-06-13","arxiv_id":"2306.08161","n_code_links":2,"syntology":null},{"paper":null,"slug":"human-like-intuitive-behavior-and-reasoning","title":"Human-Like Intuitive Behavior and Reasoning Biases Emerged in Language Models -- and Disappeared in GPT-4","date":"2023-06-13","arxiv_id":"2306.07622","n_code_links":0,"syntology":null},{"paper":"/paper/xraygpt-chest-radiographs-summarization-using","slug":"xraygpt-chest-radiographs-summarization-using","title":"XrayGPT: Chest Radiographs Summarization using Medical Vision-Language Models","date":"2023-06-13","arxiv_id":"2306.07971","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-and-non-linguistic","title":"Large language models and (non-)linguistic recursion","date":"2023-06-12","arxiv_id":"2306.07195","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-tax-attorneys-a-case","title":"Large Language Models as Tax Attorneys: A Case Study in Legal Capabilities Emergence","date":"2023-06-12","arxiv_id":"2306.07075","n_code_links":0,"syntology":null},{"paper":null,"slug":"lost-in-translation-large-language-models-in","title":"Lost in Translation: Large Language Models in Non-English Content Analysis","date":"2023-06-12","arxiv_id":"2306.07377","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-extraction-of-social","title":"Prompt-based Extraction of Social Determinants of Health Using Few-shot Learning","date":"2023-06-12","arxiv_id":"2306.07170","n_code_links":0,"syntology":null},{"paper":"/paper/inductive-reasoning-in-humans-and-large","slug":"inductive-reasoning-in-humans-and-large","title":"Inductive reasoning in humans and large language models","date":"2023-06-11","arxiv_id":"2306.06548","n_code_links":1,"syntology":null},{"paper":"/paper/14-examples-of-how-llms-can-transform","slug":"14-examples-of-how-llms-can-transform","title":"14 Examples of How LLMs Can Transform Materials Science and Chemistry: A Reflection on a Large Language Model Hackathon","date":"2023-06-09","arxiv_id":"2306.06283","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["qai222/llm_organic_synthesis","doncamilom/bollama"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","slug":"judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","arxiv_id":"2306.05685","n_code_links":11,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lm-sys/fastchat"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/m3exam-a-multilingual-multimodal-multilevel","slug":"m3exam-a-multilingual-multimodal-multilevel","title":"M3Exam: A Multilingual, Multimodal, Multilevel Benchmark for Examining Large Language Models","date":"2023-06-08","arxiv_id":"2306.05179","n_code_links":1,"syntology":null},{"paper":null,"slug":"mapping-the-challenges-of-hci-an-application","title":"Mapping the Challenges of HCI: An Application and Evaluation of ChatGPT and GPT-4 for Mining Insights at Scale","date":"2023-06-08","arxiv_id":"2306.05036","n_code_links":0,"syntology":null},{"paper":"/paper/toolalpaca-generalized-tool-learning-for","slug":"toolalpaca-generalized-tool-learning-for","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","date":"2023-06-08","arxiv_id":"2306.05301","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["tangqiaoyu/ToolAlpaca"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/good-data-large-data-or-no-data-comparing","slug":"good-data-large-data-or-no-data-comparing","title":"Good Data, Large Data, or No Data? Comparing Three Approaches in Developing Research Aspect Classifiers for Biomedical Papers","date":"2023-06-07","arxiv_id":"2306.04820","n_code_links":1,"syntology":null},{"paper":"/paper/how-far-can-camels-go-exploring-the-state-of","slug":"how-far-can-camels-go-exploring-the-state-of","title":"How Far Can Camels Go? Exploring the State of Instruction Tuning on Open Resources","date":"2023-06-07","arxiv_id":"2306.04751","n_code_links":4,"syntology":{"ran":6,"of":12,"n_ran_checked":2,"n_instrument":4,"unverified":6,"pointer_only":2,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["allenai/open-instruct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/instructeval-towards-holistic-evaluation-of","slug":"instructeval-towards-holistic-evaluation-of","title":"INSTRUCTEVAL: Towards Holistic Evaluation of Instruction-Tuned Large Language Models","date":"2023-06-07","arxiv_id":"2306.04757","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["declare-lab/instruct-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-two-word-test-a-semantic-benchmark-for","slug":"the-two-word-test-a-semantic-benchmark-for","title":"The Two Word Test: A Semantic Benchmark for Large Language Models","date":"2023-06-07","arxiv_id":"2306.04610","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/orca-progressive-learning-from-complex","slug":"orca-progressive-learning-from-complex","title":"Orca: Progressive Learning from Complex Explanation Traces of GPT-4","date":"2023-06-05","arxiv_id":"2306.02707","n_code_links":4,"syntology":null},{"paper":null,"slug":"selfevolve-a-code-evolution-framework-via","title":"SelfEvolve: A Code Evolution Framework via Large Language Models","date":"2023-06-05","arxiv_id":"2306.02907","n_code_links":0,"syntology":null},{"paper":"/paper/auto-gpt-for-online-decision-making","slug":"auto-gpt-for-online-decision-making","title":"Auto-GPT for Online Decision Making: Benchmarks and Additional Opinions","date":"2023-06-04","arxiv_id":"2306.02224","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["younghuman/llmagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-empirical-study-on-challenging-math","slug":"an-empirical-study-on-challenging-math","title":"MathChat: Converse to Tackle Challenging Math Problems with LLM Agents","date":"2023-06-02","arxiv_id":"2306.01337","n_code_links":2,"syntology":null},{"paper":null,"slug":"can-llms-like-gpt-4-outperform-traditional-ai","title":"Can LLMs like GPT-4 outperform traditional AI tools in dementia diagnosis? Maybe, but not today","date":"2023-06-02","arxiv_id":"2306.01499","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-language-models-for-mathematics","slug":"evaluating-language-models-for-mathematics","title":"Evaluating Language Models for Mathematics through Interactions","date":"2023-06-02","arxiv_id":"2306.01694","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["collinskatie/checkmate"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"hybrid-long-document-summarization-using-c2f","title":"Hybrid Long Document Summarization using C2F-FAR and ChatGPT: A Practical Study","date":"2023-06-01","arxiv_id":"2306.01169","n_code_links":0,"syntology":null},{"paper":"/paper/llava-med-training-a-large-language-and","slug":"llava-med-training-a-large-language-and","title":"LLaVA-Med: Training a Large Language-and-Vision Assistant for Biomedicine in One Day","date":"2023-06-01","arxiv_id":"2306.00890","n_code_links":1,"syntology":null},{"paper":null,"slug":"reviewergpt-an-exploratory-study-on-using","title":"ReviewerGPT? An Exploratory Study on Using Large Language Models for Paper Reviewing","date":"2023-06-01","arxiv_id":"2306.00622","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-annotation-with-generative-ai","title":"Automated Annotation with Generative AI Requires Validation","date":"2023-05-31","arxiv_id":"2306.00176","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-gpt-s-programming-capability","title":"Evaluating GPT's Programming Capability through CodeWars' Katas","date":"2023-05-31","arxiv_id":"2306.01784","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-evidence-based-instructional-design","title":"Scaling Evidence-based Instructional Design Expertise through Large Language Models","date":"2023-05-31","arxiv_id":"2306.01006","n_code_links":0,"syntology":null},{"paper":null,"slug":"alphablock-embodied-finetuning-for-vision","title":"AlphaBlock: Embodied Finetuning for Vision-Language Reasoning in Robot Manipulation","date":"2023-05-30","arxiv_id":"2305.18898","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-conceptual-representation-require","title":"Does Conceptual Representation Require Embodiment? Insights From Large Language Models","date":"2023-05-30","arxiv_id":"2305.19103","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt4geo-how-a-language-model-sees-the-world-s","title":"GPT4GEO: How a Language Model Sees the World's Geography","date":"2023-05-30","arxiv_id":"2306.00020","n_code_links":0,"syntology":null},{"paper":"/paper/gpt4tools-teaching-large-language-model-to","slug":"gpt4tools-teaching-large-language-model-to","title":"GPT4Tools: Teaching Large Language Model to Use Tools via Self-instruction","date":"2023-05-30","arxiv_id":"2305.18752","n_code_links":1,"syntology":null},{"paper":"/paper/self-verification-improves-few-shot-clinical","slug":"self-verification-improves-few-shot-clinical","title":"Self-Verification Improves Few-Shot Clinical Information Extraction","date":"2023-05-30","arxiv_id":"2306.00024","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/clinical-self-verification"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatbots-to-chatgpt-in-a-cybersecurity-space","title":"Chatbots to ChatGPT in a Cybersecurity Space: Evolution, Vulnerabilities, Attacks, Challenges, and Future Recommendations","date":"2023-05-29","arxiv_id":"2306.09255","n_code_links":0,"syntology":null},{"paper":null,"slug":"controllable-text-to-image-generation-with","title":"Controllable Text-to-Image Generation with GPT-4","date":"2023-05-29","arxiv_id":"2305.18583","n_code_links":0,"syntology":null},{"paper":"/paper/do-language-models-know-when-they-re","slug":"do-language-models-know-when-they-re","title":"Do Language Models Know When They're Hallucinating References?","date":"2023-05-29","arxiv_id":"2305.18248","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/hallucinated-references"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"game-of-tones-faculty-detection-of-gpt-4","title":"Game of Tones: Faculty detection of GPT-4 generated content in university assessments","date":"2023-05-29","arxiv_id":"2305.18081","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-not-fair-evaluators","slug":"large-language-models-are-not-fair-evaluators","title":"Large Language Models are not Fair Evaluators","date":"2023-05-29","arxiv_id":"2305.17926","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["i-eval/faireval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/marked-personas-using-natural-language","slug":"marked-personas-using-natural-language","title":"Marked Personas: Using Natural Language Prompts to Measure Stereotypes in Language Models","date":"2023-05-29","arxiv_id":"2305.18189","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["myracheng/markedpersonas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"3d8e5ed8ec69fdab7849547aed4adf97bf3684dc2def8fe2c4a2b1b551971972","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}