{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/26","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":26,"pages_in_order":29,"rows_per_page":100,"rows":[2501,2600],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/25","next":"/method/gpt-4/papers/27","papers":[{"paper":"/paper/guinea-pig-trials-utilizing-gpt-a-novel-smart","slug":"guinea-pig-trials-utilizing-gpt-a-novel-smart","title":"\"Guinea Pig Trials\" Utilizing GPT: A Novel Smart Agent-Based Modeling Approach for Studying Firm Competition and Collusion","date":"2023-08-21","arxiv_id":"2308.10974","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["roihn/sabm","wuzengqing001225/sabm_pricing_game"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-on-wikipedia-style","slug":"large-language-models-on-wikipedia-style","title":"Large Language Models on Wikipedia-Style Survey Generation: an Evaluation in NLP Concepts","date":"2023-08-21","arxiv_id":"2308.10410","n_code_links":1,"syntology":null},{"paper":"/paper/lateval-an-interactive-llms-evaluation","slug":"lateval-an-interactive-llms-evaluation","title":"LatEval: An Interactive LLMs Evaluation Benchmark with Incomplete Information from Lateral Thinking Puzzles","date":"2023-08-21","arxiv_id":"2308.10855","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":3,"n_instrument":0,"unverified":6,"pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["thukelab/lateval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-adversarial-robustness-of-multi-modal","slug":"on-the-adversarial-robustness-of-multi-modal","title":"On the Adversarial Robustness of Multi-Modal Foundation Models","date":"2023-08-21","arxiv_id":"2308.10741","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chs20/robustvlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wanjuan-a-comprehensive-multimodal-dataset","slug":"wanjuan-a-comprehensive-multimodal-dataset","title":"WanJuan: A Comprehensive Multimodal Dataset for Advancing English and Chinese Large Models","date":"2023-08-21","arxiv_id":"2308.10755","n_code_links":1,"syntology":null},{"paper":"/paper/zero-and-few-shot-prompting-with-llms-a","slug":"zero-and-few-shot-prompting-with-llms-a","title":"Zero- and Few-Shot Prompting with LLMs: A Comparative Study with Fine-tuned Models for Bangla Sentiment Analysis","date":"2023-08-21","arxiv_id":"2308.10783","n_code_links":1,"syntology":null},{"paper":"/paper/a-study-on-robustness-and-reliability-of","slug":"a-study-on-robustness-and-reliability-of","title":"Can ChatGPT replace StackOverflow? A Study on Robustness and Reliability of Large Language Model Code Generation","date":"2023-08-20","arxiv_id":"2308.10335","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["floridsleeves/robustapi"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-large-language-models-find-and-fix","title":"Can Large Language Models Find And Fix Vulnerable Software?","date":"2023-08-20","arxiv_id":"2308.10345","n_code_links":0,"syntology":null},{"paper":"/paper/chateda-a-large-language-model-powered","slug":"chateda-a-large-language-model-powered","title":"ChatEDA: A Large Language Model Powered Autonomous Agent for EDA","date":"2023-08-20","arxiv_id":"2308.10204","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["wuhy68/chatedav1"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/expel-llm-agents-are-experiential-learners","slug":"expel-llm-agents-are-experiential-learners","title":"ExpeL: LLM Agents Are Experiential Learners","date":"2023-08-20","arxiv_id":"2308.10144","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["LeapLabTHU/ExpeL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-transformers-are-better-eeg-learners","title":"Large Transformers are Better EEG Learners","date":"2023-08-20","arxiv_id":"2308.11654","n_code_links":0,"syntology":null},{"paper":"/paper/stablellava-enhanced-visual-instruction","slug":"stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","arxiv_id":"2308.10253","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["icoz69/stablellava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fineval-a-chinese-financial-domain-knowledge","slug":"fineval-a-chinese-financial-domain-knowledge","title":"FinEval: A Chinese Financial Domain Knowledge Evaluation Benchmark for Large Language Models","date":"2023-08-19","arxiv_id":"2308.09975","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sufe-aiflm-lab/fineval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-susceptible-are-llms-to-logical-fallacies","slug":"how-susceptible-are-llms-to-logical-fallacies","title":"How susceptible are LLMs to Logical Fallacies?","date":"2023-08-18","arxiv_id":"2308.09853","n_code_links":1,"syntology":null},{"paper":"/paper/red-teaming-large-language-models-using-chain","slug":"red-teaming-large-language-models-using-chain","title":"Red-Teaming Large Language Models using Chain of Utterances for Safety-Alignment","date":"2023-08-18","arxiv_id":"2308.09662","n_code_links":2,"syntology":null},{"paper":"/paper/wizardmath-empowering-mathematical-reasoning","slug":"wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","arxiv_id":"2308.09583","n_code_links":1,"syntology":{"ran":10,"of":16,"n_ran_checked":8,"n_instrument":2,"unverified":6,"pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/yorc-yoruba-reading-comprehension-dataset","slug":"yorc-yoruba-reading-comprehension-dataset","title":"NaijaRC: A Multi-choice Reading Comprehension Dataset for Nigerian Languages","date":"2023-08-18","arxiv_id":"2308.09768","n_code_links":1,"syntology":null},{"paper":"/paper/chat-3d-data-efficiently-tuning-large","slug":"chat-3d-data-efficiently-tuning-large","title":"Chat-3D: Data-efficiently Tuning Large Language Model for Universal Dialogue of 3D Scenes","date":"2023-08-17","arxiv_id":"2308.08769","n_code_links":2,"syntology":null},{"paper":"/paper/cmb-a-comprehensive-medical-benchmark-in","slug":"cmb-a-comprehensive-medical-benchmark-in","title":"CMB: A Comprehensive Medical Benchmark in Chinese","date":"2023-08-17","arxiv_id":"2308.08833","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["FreedomIntelligence/CMB"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"mascqa-a-question-answering-dataset-for","title":"MaScQA: A Question Answering Dataset for Investigating Materials Science Knowledge of Large Language Models","date":"2023-08-17","arxiv_id":"2308.09115","n_code_links":0,"syntology":null},{"paper":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"boosting-logical-reasoning-in-large-language","title":"Boosting Logical Reasoning in Large Language Models through a New Framework: The Graph of Thought","date":"2023-08-16","arxiv_id":"2308.08614","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-deception-reverse-penetrating-the","title":"Self-Deception: Reverse Penetrating the Semantic Firewall of Large Language Models","date":"2023-08-16","arxiv_id":"2308.11521","n_code_links":0,"syntology":null},{"paper":"/paper/time-travel-in-llms-tracing-data","slug":"time-travel-in-llms-tracing-data","title":"Time Travel in LLMs: Tracing Data Contamination in Large Language Models","date":"2023-08-16","arxiv_id":"2308.08493","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["shahriargolchin/time-travel-in-llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automated-test-case-generation-using-code","title":"Domain Adaptation for Code Model-based Unit Test Case Generation","date":"2023-08-15","arxiv_id":"2308.08033","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-introductory","title":"Large Language Models in Introductory Programming Education: ChatGPT's Performance and Implications for Assessments","date":"2023-08-15","arxiv_id":"2308.08572","n_code_links":0,"syntology":null},{"paper":"/paper/solving-challenging-math-word-problems-using","slug":"solving-challenging-math-word-problems-using","title":"Solving Challenging Math Word Problems Using GPT-4 Code Interpreter with Code-based Self-Verification","date":"2023-08-15","arxiv_id":"2308.07921","n_code_links":1,"syntology":null},{"paper":null,"slug":"chatbots-in-drug-discovery-a-case-study-on","title":"ChatGPT in Drug Discovery: A Case Study on Anti-Cocaine Addiction Drug Development with Chatbots","date":"2023-08-14","arxiv_id":"2308.06920","n_code_links":0,"syntology":null},{"paper":"/paper/dialogue-for-prompting-a-policy-gradient","slug":"dialogue-for-prompting-a-policy-gradient","title":"Dialogue for Prompting: a Policy-Gradient-Based Discrete Prompt Generation for Few-shot Learning","date":"2023-08-14","arxiv_id":"2308.07272","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["czx-li/DP2O"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-for-information","slug":"large-language-models-for-information","title":"Large Language Models for Information Retrieval: A Survey","date":"2023-08-14","arxiv_id":"2308.07107","n_code_links":1,"syntology":null},{"paper":"/paper/neural-authorship-attribution-stylometric","slug":"neural-authorship-attribution-stylometric","title":"Neural Authorship Attribution: Stylometric Analysis on Large Language Models","date":"2023-08-14","arxiv_id":"2308.07305","n_code_links":1,"syntology":null},{"paper":"/paper/gpt-4-is-too-smart-to-be-safe-stealthy-chat","slug":"gpt-4-is-too-smart-to-be-safe-stealthy-chat","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","date":"2023-08-12","arxiv_id":"2308.06463","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["robustnlp/cipherchat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/visit-bench-a-benchmark-for-vision-language","slug":"visit-bench-a-benchmark-for-vision-language","title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","date":"2023-08-12","arxiv_id":"2308.06595","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":3,"n_instrument":6,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mlfoundations/VisIT-Bench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-student-errors-in-experimentation","title":"Assessing Student Errors in Experimentation Using Artificial Intelligence and Large Language Models: A Comparative Study with Human Raters","date":"2023-08-11","arxiv_id":"2308.06088","n_code_links":0,"syntology":null},{"paper":"/paper/learning-deductive-reasoning-from-synthetic","slug":"learning-deductive-reasoning-from-synthetic","title":"Learning Deductive Reasoning from Synthetic Corpus based on Formal Logic","date":"2023-08-11","arxiv_id":"2308.07336","n_code_links":3,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hitachi-nlp/fld"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/adaptive-low-rank-adaptation-of-segment","slug":"adaptive-low-rank-adaptation-of-segment","title":"Adaptive Low Rank Adaptation of Segment Anything to Salient Object Detection","date":"2023-08-10","arxiv_id":"2308.05426","n_code_links":1,"syntology":null},{"paper":"/paper/metacognitive-prompting-improves","slug":"metacognitive-prompting-improves","title":"Metacognitive Prompting Improves Understanding in Large Language Models","date":"2023-08-10","arxiv_id":"2308.05342","n_code_links":1,"syntology":null},{"paper":null,"slug":"testing-gpt-4-with-wolfram-alpha-and-code","title":"Testing GPT-4 with Wolfram Alpha and Code Interpreter plug-ins on math and science problems","date":"2023-08-10","arxiv_id":"2308.05713","n_code_links":0,"syntology":null},{"paper":"/paper/trustworthy-llms-a-survey-and-guideline-for","slug":"trustworthy-llms-a-survey-and-guideline-for","title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","date":"2023-08-10","arxiv_id":"2308.05374","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-open-source-large","title":"A Comparative Study of Open-Source Large Language Models, GPT-4 and Claude 2: Multiple-Choice Test Taking in Nephrology","date":"2023-08-09","arxiv_id":"2308.04709","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-for-arabic-grammatical-error","title":"ChatGPT for Arabic Grammatical Error Correction","date":"2023-08-08","arxiv_id":"2308.04492","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-color-similarity-structures-between","title":"Gromov-Wasserstein unsupervised alignment reveals structural correspondences between the color similarity structures of humans and large language models","date":"2023-08-08","arxiv_id":"2308.04381","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-medical-image-classification-with","title":"Few-shot medical image classification with simple shape and texture text descriptors using vision-language models","date":"2023-08-08","arxiv_id":"2308.04005","n_code_links":0,"syntology":null},{"paper":"/paper/shepherd-a-critic-for-language-model","slug":"shepherd-a-critic-for-language-model","title":"Shepherd: A Critic for Language Model Generation","date":"2023-08-08","arxiv_id":"2308.04592","n_code_links":1,"syntology":null},{"paper":"/paper/do-anything-now-characterizing-and-evaluating","slug":"do-anything-now-characterizing-and-evaluating","title":"\"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models","date":"2023-08-07","arxiv_id":"2308.03825","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["verazuo/jailbreak_llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/emotionally-numb-or-empathetic-evaluating-how","slug":"emotionally-numb-or-empathetic-evaluating-how","title":"Emotionally Numb or Empathetic? Evaluating How LLMs Feel Using EmotionBench","date":"2023-08-07","arxiv_id":"2308.03656","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cuhk-arise/emotionbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/extracting-detailed-oncologic-history-and","slug":"extracting-detailed-oncologic-history-and","title":"CORAL: Expert-Curated medical Oncology Reports to Advance Language Model Inference","date":"2023-08-07","arxiv_id":"2308.03853","n_code_links":1,"syntology":null},{"paper":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"symmetry-preserving-program-representations","title":"Exploiting Code Symmetries for Learning Program Semantics","date":"2023-08-07","arxiv_id":"2308.03312","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-trained-large-language-models-for","title":"Pre-Trained Large Language Models for Industrial Control","date":"2023-08-06","arxiv_id":"2308.03028","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-criterion-for-artificial-general","title":"A criterion for Artificial General Intelligence: hypothetic-deductive reasoning, tested on ChatGPT","date":"2023-08-05","arxiv_id":"2308.02950","n_code_links":0,"syntology":null},{"paper":"/paper/chatgpt-for-gtfs-from-words-to-information","slug":"chatgpt-for-gtfs-from-words-to-information","title":"ChatGPT for GTFS: Benchmarking LLMs on GTFS Understanding and Retrieval","date":"2023-08-04","arxiv_id":"2308.02618","n_code_links":1,"syntology":null},{"paper":null,"slug":"scaling-clinical-trial-matching-using-large","title":"Scaling Clinical Trial Matching Using Large Language Models: A Case Study in Oncology","date":"2023-08-04","arxiv_id":"2308.02180","n_code_links":0,"syntology":null},{"paper":"/paper/classeval-a-manually-crafted-benchmark-for","slug":"classeval-a-manually-crafted-benchmark-for","title":"ClassEval: A Manually-Crafted Benchmark for Evaluating LLMs on Class-level Code Generation","date":"2023-08-03","arxiv_id":"2308.01861","n_code_links":2,"syntology":null},{"paper":null,"slug":"is-gpt-4-a-reliable-rater-evaluating","title":"Is GPT-4 a reliable rater? Evaluating Consistency in GPT-4 Text Ratings","date":"2023-08-03","arxiv_id":"2308.02575","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-displays-emergent","title":"Large Language Model Displays Emergent Ability to Interpret Novel Literary Metaphors","date":"2023-08-03","arxiv_id":"2308.01497","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-psychology-of-gpt-4-s-moral-and","title":"Exploring the psychology of LLMs' Moral and Legal Reasoning","date":"2023-08-02","arxiv_id":"2308.01264","n_code_links":0,"syntology":null},{"paper":"/paper/flows-building-blocks-of-reasoning-and","slug":"flows-building-blocks-of-reasoning-and","title":"Flows: Building Blocks of Reasoning and Collaborating AI","date":"2023-08-02","arxiv_id":"2308.01285","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["epfl-dlab/aiflows","epfl-dlab/cc_flows"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-effective-data-creation-pipeline-to","title":"An Effective Data Creation Pipeline to Generate High-quality Financial Instruction Data for Large Language Model","date":"2023-07-31","arxiv_id":"2308.01415","n_code_links":0,"syntology":null},{"paper":null,"slug":"deception-abilities-emerged-in-large-language","title":"Deception Abilities Emerged in Large Language Models","date":"2023-07-31","arxiv_id":"2307.16513","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-chatgpt-and-gpt-4-for-visual","title":"Evaluating ChatGPT and GPT-4 for Visual Programming","date":"2023-07-30","arxiv_id":"2308.02522","n_code_links":0,"syntology":null},{"paper":"/paper/chathome-development-and-evaluation-of-a","slug":"chathome-development-and-evaluation-of-a","title":"ChatHome: Development and Evaluation of a Domain-Specific Language Model for Home Renovation","date":"2023-07-28","arxiv_id":"2307.15290","n_code_links":1,"syntology":{"ran":7,"of":17,"n_ran_checked":4,"n_instrument":3,"unverified":10,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","official":{"repos":["lianjiatech/belle"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":10,"ran_from_kinds":["official"]}}},{"paper":"/paper/rsgpt-a-remote-sensing-vision-language-model","slug":"rsgpt-a-remote-sensing-vision-language-model","title":"RSGPT: A Remote Sensing Vision Language Model and Benchmark","date":"2023-07-28","arxiv_id":"2307.15266","n_code_links":2,"syntology":null},{"paper":null,"slug":"llmediator-gpt-4-assisted-online-dispute","title":"LLMediator: GPT-4 Assisted Online Dispute Resolution","date":"2023-07-27","arxiv_id":"2307.16732","n_code_links":0,"syntology":null},{"paper":"/paper/superclue-a-comprehensive-chinese-large","slug":"superclue-a-comprehensive-chinese-large","title":"SuperCLUE: A Comprehensive Chinese Large Language Model Benchmark","date":"2023-07-27","arxiv_id":"2307.15020","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-large-language-models-for-mental","slug":"leveraging-large-language-models-for-mental","title":"Mental-LLM: Leveraging Large Language Models for Mental Health Prediction via Online Text Data","date":"2023-07-26","arxiv_id":"2307.14385","n_code_links":1,"syntology":null},{"paper":null,"slug":"unveiling-security-privacy-and-ethical","title":"Unveiling Security, Privacy, and Ethical Concerns of ChatGPT","date":"2023-07-26","arxiv_id":"2307.14192","n_code_links":0,"syntology":null},{"paper":null,"slug":"arb-advanced-reasoning-benchmark-for-large","title":"ARB: Advanced Reasoning Benchmark for Large Language Models","date":"2023-07-25","arxiv_id":"2307.13692","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-can-large-language-models-help-humans-in","title":"How Can Large Language Models Help Humans in Design and Manufacturing?","date":"2023-07-25","arxiv_id":"2307.14377","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-gpt-a-computational-model-of-emotion","title":"Is GPT a Computational Model of Emotion? Detailed Analysis","date":"2023-07-25","arxiv_id":"2307.13779","n_code_links":0,"syntology":null},{"paper":"/paper/predicting-code-coverage-without-execution","slug":"predicting-code-coverage-without-execution","title":"Predicting Code Coverage without Execution","date":"2023-07-25","arxiv_id":"2307.13383","n_code_links":1,"syntology":null},{"paper":null,"slug":"performance-of-large-language-models-in-a","title":"Performance of Large Language Models in a Computer Science Degree Program","date":"2023-07-24","arxiv_id":"2308.02432","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-clip-with-gpt-4-harnessing-visual","slug":"enhancing-clip-with-gpt-4-harnessing-visual","title":"Enhancing CLIP with GPT-4: Harnessing Visual Descriptions as Prompts","date":"2023-07-21","arxiv_id":"2307.11661","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mayug/vdt-adapter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-4-can-t-reason","title":"GPT-4 Can't Reason","date":"2023-07-21","arxiv_id":"2308.03762","n_code_links":0,"syntology":null},{"paper":null,"slug":"predict-ai-bility-of-how-humans-balance-self","title":"Assessing Large Language Models' ability to predict how humans balance self-interest and the interest of others","date":"2023-07-21","arxiv_id":"2307.12776","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-llm-assisted-exploitation-of-ai-guardian","title":"A LLM Assisted Exploitation of AI-Guardian","date":"2023-07-20","arxiv_id":"2307.15008","n_code_links":0,"syntology":null},{"paper":null,"slug":"instruction-following-evaluation-through","title":"Instruction-following Evaluation through Verbalizer Manipulation","date":"2023-07-20","arxiv_id":"2307.10558","n_code_links":0,"syntology":null},{"paper":"/paper/l-eval-instituting-standardized-evaluation","slug":"l-eval-instituting-standardized-evaluation","title":"L-Eval: Instituting Standardized Evaluation for Long Context Language Models","date":"2023-07-20","arxiv_id":"2307.11088","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["openlmlab/leval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/of-models-and-tin-men-a-behavioural-economics","slug":"of-models-and-tin-men-a-behavioural-economics","title":"Of Models and Tin Men: A Behavioural Economics Study of Principal-Agent Problems in AI Alignment using Large-Language Models","date":"2023-07-20","arxiv_id":"2307.11137","n_code_links":2,"syntology":null},{"paper":null,"slug":"pharmacygpt-the-ai-pharmacist","title":"PharmacyGPT: The AI Pharmacist","date":"2023-07-19","arxiv_id":"2307.10432","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatspot-bootstrapping-multimodal-llms-via","title":"ChatSpot: Bootstrapping Multimodal LLMs via Precise Referring Instruction Tuning","date":"2023-07-18","arxiv_id":"2307.09474","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotional-intelligence-of-large-language","title":"Emotional Intelligence of Large Language Models","date":"2023-07-18","arxiv_id":"2307.09042","n_code_links":0,"syntology":null},{"paper":"/paper/how-is-chatgpt-s-behavior-changing-over-time","slug":"how-is-chatgpt-s-behavior-changing-over-time","title":"How is ChatGPT's behavior changing over time?","date":"2023-07-18","arxiv_id":"2307.09009","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lchen001/llmdrift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"abductive-reasoning-with-the-gpt-4-language","title":"Abductive Reasoning with the GPT-4 Language Model: Case studies from criminal investigation, medical practice, scientific research","date":"2023-07-17","arxiv_id":"2307.10250","n_code_links":0,"syntology":null},{"paper":"/paper/alpagasus-training-a-better-alpaca-with-fewer","slug":"alpagasus-training-a-better-alpaca-with-fewer","title":"AlpaGasus: Training A Better Alpaca with Fewer Data","date":"2023-07-17","arxiv_id":"2307.08701","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gpt4life/alpagasus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatgpt-is-good-but-bing-chat-is-better-for","title":"ChatGPT is Good but Bing Chat is Better for Vietnamese Students","date":"2023-07-17","arxiv_id":"2307.08272","n_code_links":0,"syntology":null},{"paper":"/paper/collie-systematic-construction-of-constrained","slug":"collie-systematic-construction-of-constrained","title":"COLLIE: Systematic Construction of Constrained Text Generation Tasks","date":"2023-07-17","arxiv_id":"2307.08689","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["princeton-nlp/Collie"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"in-ide-generation-based-information-support","title":"Using an LLM to Help With Code Understanding","date":"2023-07-17","arxiv_id":"2307.08177","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-application-of-large-language-models","title":"On the application of Large Language Models for language teaching and assessment technology","date":"2023-07-17","arxiv_id":"2307.08393","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-the-quality-of-multiple-choice","slug":"assessing-the-quality-of-multiple-choice","title":"Assessing the Quality of Multiple-Choice Questions Using GPT-4 and Rule-Based Methods","date":"2023-07-16","arxiv_id":"2307.08161","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-potential-and-pitfalls-of-using-a-large","title":"The Potential and Pitfalls of using a Large Language Model such as ChatGPT or GPT-4 as a Clinical Assistant","date":"2023-07-16","arxiv_id":"2307.08152","n_code_links":0,"syntology":null},{"paper":"/paper/creating-a-dataset-supporting-translation","slug":"creating-a-dataset-supporting-translation","title":"Creating a Dataset for High-Performance Computing Code Translation using LLMs: A Bridge Between OpenMP Fortran and C++","date":"2023-07-15","arxiv_id":"2307.07686","n_code_links":1,"syntology":null},{"paper":"/paper/leveraging-large-language-models-to-generate","slug":"leveraging-large-language-models-to-generate","title":"Leveraging Large Language Models to Generate Answer Set Programs","date":"2023-07-15","arxiv_id":"2307.07699","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["azreasoners/gpt-asp-rules"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/think-on-graph-deep-and-responsible-reasoning","slug":"think-on-graph-deep-and-responsible-reasoning","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","date":"2023-07-15","arxiv_id":"2307.07697","n_code_links":3,"syntology":{"ran":12,"of":16,"n_ran_checked":10,"n_instrument":2,"unverified":4,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gasolsun36/tog","idea-finai/tog"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"emotionprompt-leveraging-psychology-for-large","title":"Large Language Models Understand and Can be Enhanced by Emotional Stimuli","date":"2023-07-14","arxiv_id":"2307.11760","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-integration-of-large-language","title":"Exploring the Integration of Large Language Models into Automatic Speech Recognition Systems: An Empirical Study","date":"2023-07-13","arxiv_id":"2307.06530","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for","title":"Distilling Large Language Models for Biomedical Knowledge Extraction: A Case Study on Adverse Drug Events","date":"2023-07-12","arxiv_id":"2307.06439","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-generate-train-pgt-a-framework-for-few","title":"Prompt Generate Train (PGT): Few-shot Domain Adaption of Retrieval Augmented Generation Models for Open Book Question-Answering","date":"2023-07-12","arxiv_id":"2307.05915","n_code_links":0,"syntology":null},{"paper":null,"slug":"argumentative-segmentation-enhancement-for","title":"Argumentative Segmentation Enhancement for Legal Summarization","date":"2023-07-11","arxiv_id":"2307.05081","n_code_links":0,"syntology":null},{"paper":null,"slug":"explaining-competitive-level-programming","title":"Explaining Competitive-Level Programming Solutions using LLMs","date":"2023-07-11","arxiv_id":"2307.05337","n_code_links":0,"syntology":null}],"record_sha256":"1f19afe6b4fbe869e8d5c2d9347fbf0a13de7b9142b7942e12dda137261f03fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}