{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/23","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":23,"pages_in_order":29,"rows_per_page":100,"rows":[2201,2300],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/22","next":"/method/gpt-4/papers/24","papers":[{"paper":null,"slug":"building-real-world-meeting-summarization","title":"Building Real-World Meeting Summarization Systems using Large Language Models: A Practical Perspective","date":"2023-10-30","arxiv_id":"2310.19233","n_code_links":0,"syntology":null},{"paper":null,"slug":"constituency-parsing-using-llms","title":"Constituency Parsing using LLMs","date":"2023-10-30","arxiv_id":"2310.19462","n_code_links":0,"syntology":null},{"paper":"/paper/dynamics-of-instruction-tuning-each-ability","slug":"dynamics-of-instruction-tuning-each-ability","title":"Dynamics of Instruction Tuning: Each Ability of Large Language Models Has Its Own Growth Pace","date":"2023-10-30","arxiv_id":"2310.19651","n_code_links":1,"syntology":null},{"paper":"/paper/interpretable-by-design-text-classification","slug":"interpretable-by-design-text-classification","title":"Interpretable-by-Design Text Understanding with Iteratively Generated Concept Bottleneck","date":"2023-10-30","arxiv_id":"2310.19660","n_code_links":1,"syntology":null},{"paper":"/paper/multimodal-chatgpt-for-medical-applications","slug":"multimodal-chatgpt-for-medical-applications","title":"Multimodal ChatGPT for Medical Applications: an Experimental Study of GPT-4V","date":"2023-10-29","arxiv_id":"2310.19061","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-large-language-models-to-support","title":"Using Large Language Models to Support Thematic Analysis in Empirical Legal Studies","date":"2023-10-28","arxiv_id":"2310.18729","n_code_links":0,"syntology":null},{"paper":"/paper/can-llms-keep-a-secret-testing-privacy","slug":"can-llms-keep-a-secret-testing-privacy","title":"Can LLMs Keep a Secret? Testing Privacy Implications of Language Models via Contextual Integrity Theory","date":"2023-10-27","arxiv_id":"2310.17884","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-4-vision-on-medical-image-classification","title":"GPT-4 Vision on Medical Image Classification -- A Case Study on COVID-19 Dataset","date":"2023-10-27","arxiv_id":"2310.18498","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowing-what-llms-do-not-know-a-simple-yet","title":"Knowing What LLMs DO NOT Know: A Simple Yet Effective Self-Detection Method","date":"2023-10-27","arxiv_id":"2310.17918","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-aspect-based","slug":"large-language-models-for-aspect-based","title":"Large language models for aspect-based sentiment analysis","date":"2023-10-27","arxiv_id":"2310.18025","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qagentur/absa_llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/soul-towards-sentiment-and-opinion","slug":"soul-towards-sentiment-and-opinion","title":"SOUL: Towards Sentiment and Opinion Understanding of Language","date":"2023-10-27","arxiv_id":"2310.17924","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["damo-nlp-sg/soul"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-framework-for-automated-measurement-of","title":"A Framework for Automated Measurement of Responsible AI Harms in Generative AI Applications","date":"2023-10-26","arxiv_id":"2310.17750","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-replace-humans-in","title":"Can large language models replace humans in the systematic review process? Evaluating GPT-4's efficacy in screening and extracting data from peer-reviewed and grey literature in multiple languages","date":"2023-10-26","arxiv_id":"2310.17526","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-grade-short-answer-reading","title":"Can LLMs Grade Short-Answer Reading Comprehension Questions : An Empirical Study with a Novel Dataset","date":"2023-10-26","arxiv_id":"2310.18373","n_code_links":0,"syntology":null},{"paper":"/paper/competeai-understanding-the-competition","slug":"competeai-understanding-the-competition","title":"CompeteAI: Understanding the Competition Dynamics in Large Language Model-based Agents","date":"2023-10-26","arxiv_id":"2310.17512","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["microsoft/competeai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cultural-adaptation-of-recipes","title":"Cultural Adaptation of Recipes","date":"2023-10-26","arxiv_id":"2310.17353","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-explanation-the-cure-misinformation","title":"Is Explanation the Cure? Misinformation Mitigation in the Short Term and Long Term","date":"2023-10-26","arxiv_id":"2310.17711","n_code_links":0,"syntology":null},{"paper":null,"slug":"skill-mix-a-flexible-and-expandable-family-of","title":"Skill-Mix: a Flexible and Expandable Family of Evaluations for AI models","date":"2023-10-26","arxiv_id":"2310.17567","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-are-an-expert-linguistic-annotator-limits","title":"\"You Are An Expert Linguistic Annotator\": Limits of LLMs as Analyzers of Abstract Meaning Representation","date":"2023-10-26","arxiv_id":"2310.17793","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comprehensive-evaluation-of-constrained","title":"Evaluating, Understanding, and Improving Constrained Text Generation for Large Language Models","date":"2023-10-25","arxiv_id":"2310.16343","n_code_links":0,"syntology":null},{"paper":"/paper/an-early-evaluation-of-gpt-4v-ision","slug":"an-early-evaluation-of-gpt-4v-ision","title":"An Early Evaluation of GPT-4V(ision)","date":"2023-10-25","arxiv_id":"2310.16534","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-models-follow-human-summarization","title":"Can GPT models Follow Human Summarization Guidelines? Evaluating ChatGPT and GPT-4 for Dialogue Summarization","date":"2023-10-25","arxiv_id":"2310.16810","n_code_links":0,"syntology":null},{"paper":"/paper/is-chatgpt-a-good-multi-party-conversation","slug":"is-chatgpt-a-good-multi-party-conversation","title":"Is ChatGPT a Good Multi-Party Conversation Solver?","date":"2023-10-25","arxiv_id":"2310.16301","n_code_links":1,"syntology":null},{"paper":"/paper/llm-performance-predictors-are-good","slug":"llm-performance-predictors-are-good","title":"LLM Performance Predictors are good initializers for Architecture Search","date":"2023-10-25","arxiv_id":"2310.16712","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ubc-nlp/llmas"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/netfound-foundation-model-for-network","slug":"netfound-foundation-model-for-network","title":"netFound: Foundation Model for Network Security","date":"2023-10-25","arxiv_id":"2310.17025","n_code_links":1,"syntology":null},{"paper":"/paper/occuquest-mitigating-occupational-bias-for","slug":"occuquest-mitigating-occupational-bias-for","title":"OccuQuest: Mitigating Occupational Bias for Inclusive Large Language Models","date":"2023-10-25","arxiv_id":"2310.16517","n_code_links":1,"syntology":null},{"paper":"/paper/superhf-supervised-iterative-learning-from","slug":"superhf-supervised-iterative-learning-from","title":"SuperHF: Supervised Iterative Learning from Human Feedback","date":"2023-10-25","arxiv_id":"2310.16763","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["openfeedback/superhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"using-gpt-4-to-augment-unbalanced-data-for","title":"Using GPT-4 to Augment Unbalanced Data for Automatic Scoring","date":"2023-10-25","arxiv_id":"2310.18365","n_code_links":0,"syntology":null},{"paper":null,"slug":"confounder-balancing-in-adversarial-domain","title":"Confounder Balancing in Adversarial Domain Adaptation for Pre-Trained Large Models Fine-Tuning","date":"2023-10-24","arxiv_id":"2310.16062","n_code_links":0,"syntology":null},{"paper":"/paper/musr-testing-the-limits-of-chain-of-thought","slug":"musr-testing-the-limits-of-chain-of-thought","title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","date":"2023-10-24","arxiv_id":"2310.16049","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zayne-sprague/musr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/notechat-a-dataset-of-synthetic-doctor","slug":"notechat-a-dataset-of-synthetic-doctor","title":"NoteChat: A Dataset of Synthetic Doctor-Patient Conversations Conditioned on Clinical Notes","date":"2023-10-24","arxiv_id":"2310.15959","n_code_links":1,"syntology":null},{"paper":null,"slug":"ui-layout-generation-with-llms-guided-by-ui","title":"UI Layout Generation with LLMs Guided by UI Grammar","date":"2023-10-24","arxiv_id":"2310.15455","n_code_links":0,"syntology":null},{"paper":"/paper/alpacare-instruction-tuned-large-language","slug":"alpacare-instruction-tuned-large-language","title":"AlpaCare:Instruction-tuned Large Language Models for Medical Application","date":"2023-10-23","arxiv_id":"2310.14558","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xzhang97666/alpacare"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analyzing-multilingual-competency-of-llms-in","title":"Analyzing Multilingual Competency of LLMs in Multi-Turn Instruction Following: A Case Study of Arabic","date":"2023-10-23","arxiv_id":"2310.14819","n_code_links":0,"syntology":null},{"paper":null,"slug":"branch-solve-merge-improves-large-language","title":"Branch-Solve-Merge Improves Large Language Model Evaluation and Generation","date":"2023-10-23","arxiv_id":"2310.15123","n_code_links":0,"syntology":null},{"paper":null,"slug":"causal-inference-using-llm-guided-discovery","title":"Causal Inference Using LLM-Guided Discovery","date":"2023-10-23","arxiv_id":"2310.15117","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-spatial-understanding-of-large","slug":"evaluating-spatial-understanding-of-large","title":"Evaluating Spatial Understanding of Large Language Models","date":"2023-10-23","arxiv_id":"2310.14540","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["runopti/spatialevalllm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-the-knowledge-base-completion","title":"Evaluating the Knowledge Base Completion Potential of GPT","date":"2023-10-23","arxiv_id":"2310.14771","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-boundaries-of-gpt-4-in","title":"Exploring the Boundaries of GPT-4 in Radiology","date":"2023-10-23","arxiv_id":"2310.14573","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-as-an-effective-zero-shot-evaluator-for","title":"GPT-4 as an Effective Zero-Shot Evaluator for Scientific Figure Captions","date":"2023-10-23","arxiv_id":"2310.15405","n_code_links":0,"syntology":null},{"paper":null,"slug":"instructexcel-a-benchmark-for-natural","title":"InstructExcel: A Benchmark for Natural Language Instruction in Excel","date":"2023-10-23","arxiv_id":"2310.14495","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-can-share-images-too","slug":"large-language-models-can-share-images-too","title":"Large Language Models can Share Images, Too!","date":"2023-10-23","arxiv_id":"2310.14804","n_code_links":2,"syntology":null},{"paper":"/paper/linc-a-neurosymbolic-approach-for-logical","slug":"linc-a-neurosymbolic-approach-for-logical","title":"LINC: A Neurosymbolic Approach for Logical Reasoning by Combining Language Models with First-Order Logic Provers","date":"2023-10-23","arxiv_id":"2310.15164","n_code_links":1,"syntology":{"ran":1,"of":7,"n_ran_checked":0,"n_instrument":1,"unverified":6,"pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["benlipkin/linc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/location-aware-visual-question-generation","slug":"location-aware-visual-question-generation","title":"Location-Aware Visual Question Generation with Lightweight Models","date":"2023-10-23","arxiv_id":"2310.15129","n_code_links":1,"syntology":null},{"paper":"/paper/teleqna-a-benchmark-dataset-to-assess-large","slug":"teleqna-a-benchmark-dataset-to-assess-large","title":"TeleQnA: A Benchmark Dataset to Assess Large Language Models Telecommunications Knowledge","date":"2023-10-23","arxiv_id":"2310.15051","n_code_links":1,"syntology":null},{"paper":null,"slug":"unleashing-the-potential-of-prompt","title":"Unleashing the potential of prompt engineering for large language models","date":"2023-10-23","arxiv_id":"2310.14735","n_code_links":0,"syntology":null},{"paper":"/paper/cxr-llava-multimodal-large-language-model-for","slug":"cxr-llava-multimodal-large-language-model-for","title":"CXR-LLAVA: a multimodal large language model for interpreting chest X-ray images","date":"2023-10-22","arxiv_id":"2310.18341","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ecofri/cxr_llava"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-are-biased-to","slug":"large-language-models-are-biased-to","title":"Large Language Models are biased to overestimate profoundness","date":"2023-10-22","arxiv_id":"2310.14422","n_code_links":1,"syntology":null},{"paper":"/paper/gemba-mqm-detecting-translation-quality-error","slug":"gemba-mqm-detecting-translation-quality-error","title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","date":"2023-10-21","arxiv_id":"2310.13988","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/small-language-models-fine-tuned-to","slug":"small-language-models-fine-tuned-to","title":"Small Language Models Fine-tuned to Coordinate Larger Language Models improve Complex Reasoning","date":"2023-10-21","arxiv_id":"2310.18338","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lcs2-iiitd/daslam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"alltogether-investigating-the-efficacy-of","title":"AllTogether: Investigating the Efficacy of Spliced Prompt for Web Navigation using Large Language Models","date":"2023-10-20","arxiv_id":"2310.18331","n_code_links":0,"syntology":null},{"paper":null,"slug":"ask-language-model-to-clean-your-noisy","title":"Ask Language Model to Clean Your Noisy Translation Data","date":"2023-10-20","arxiv_id":"2310.13469","n_code_links":0,"syntology":null},{"paper":"/paper/botchat-evaluating-llms-capabilities-of","slug":"botchat-evaluating-llms-capabilities-of","title":"BotChat: Evaluating LLMs' Capabilities of Having Multi-Turn Dialogues","date":"2023-10-20","arxiv_id":"2310.13650","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["open-compass/botchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cache-me-if-you-can-an-online-cost-aware","slug":"cache-me-if-you-can-an-online-cost-aware","title":"Cache me if you Can: an Online Cost-aware Teacher-Student framework to Reduce the Calls to Large Language Models","date":"2023-10-20","arxiv_id":"2310.13395","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stoyian/OCaTS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"design-inclusive-language-models-for","title":"She had Cobalt Blue Eyes: Prompt Testing to Create Aligned and Sustainable Language Models","date":"2023-10-20","arxiv_id":"2310.18333","n_code_links":0,"syntology":null},{"paper":"/paper/evaluation-metrics-in-the-era-of-gpt-4","slug":"evaluation-metrics-in-the-era-of-gpt-4","title":"Evaluation Metrics in the Era of GPT-4: Reliably Evaluating Large Language Models on Sequence to Sequence Tasks","date":"2023-10-20","arxiv_id":"2310.13800","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["protagolabs/seq2seq_llm_evaluation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/moqagpt-zero-shot-multi-modal-open-domain","slug":"moqagpt-zero-shot-multi-modal-open-domain","title":"MoqaGPT : Zero-Shot Multi-modal Open-domain Question Answering with Large Language Model","date":"2023-10-20","arxiv_id":"2310.13265","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-perils-promises-of-fact-checking-with","title":"The Perils & Promises of Fact-checking with Large Language Models","date":"2023-10-20","arxiv_id":"2310.13549","n_code_links":0,"syntology":null},{"paper":"/paper/tuna-instruction-tuning-using-feedback-from","slug":"tuna-instruction-tuning-using-feedback-from","title":"Tuna: Instruction Tuning using Feedback from Large Language Models","date":"2023-10-20","arxiv_id":"2310.13385","n_code_links":1,"syntology":null},{"paper":"/paper/agenttuning-enabling-generalized-agent","slug":"agenttuning-enabling-generalized-agent","title":"AgentTuning: Enabling Generalized Agent Abilities for LLMs","date":"2023-10-19","arxiv_id":"2310.12823","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thudm/agenttuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automatic-hallucination-assessment-for","title":"ReEval: Automatic Hallucination Evaluation for Retrieval-Augmented Large Language Models via Transferable Adversarial Attacks","date":"2023-10-19","arxiv_id":"2310.12516","n_code_links":0,"syntology":null},{"paper":"/paper/automix-automatically-mixing-language-models","slug":"automix-automatically-mixing-language-models","title":"AutoMix: Automatically Mixing Language Models","date":"2023-10-19","arxiv_id":"2310.12963","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["automix-llm/automix"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":9,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"paper":null,"slug":"experimental-narratives-a-comparison-of-human","title":"Experimental Narratives: A Comparison of Human Crowdsourced Storytelling and AI Storytelling","date":"2023-10-19","arxiv_id":"2310.12902","n_code_links":0,"syntology":null},{"paper":null,"slug":"not-all-countries-celebrate-thanksgiving-on","title":"Not All Countries Celebrate Thanksgiving: On the Cultural Dominance in Large Language Models","date":"2023-10-19","arxiv_id":"2310.12481","n_code_links":0,"syntology":null},{"paper":"/paper/product-attribute-value-extraction-using","slug":"product-attribute-value-extraction-using","title":"ExtractGPT: Exploring the Potential of Large Language Models for Product Attribute Value Extraction","date":"2023-10-19","arxiv_id":"2310.12537","n_code_links":1,"syntology":null},{"paper":"/paper/the-foundation-model-transparency-index","slug":"the-foundation-model-transparency-index","title":"The Foundation Model Transparency Index","date":"2023-10-19","arxiv_id":"2310.12941","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-symbol-binding-ability-of","title":"Evaluating the Symbol Binding Ability of Large Language Models for Multiple-Choice Questions in Vietnamese General Education","date":"2023-10-18","arxiv_id":"2310.12059","n_code_links":0,"syntology":null},{"paper":"/paper/quantify-health-related-atomic-knowledge-in","slug":"quantify-health-related-atomic-knowledge-in","title":"Quantifying Self-diagnostic Atomic Knowledge in Chinese Medical Foundation Model: A Computational Analysis","date":"2023-10-18","arxiv_id":"2310.11722","n_code_links":1,"syntology":null},{"paper":"/paper/sotopia-interactive-evaluation-for-social","slug":"sotopia-interactive-evaluation-for-social","title":"SOTOPIA: Interactive Evaluation for Social Intelligence in Language Agents","date":"2023-10-18","arxiv_id":"2310.11667","n_code_links":2,"syntology":null},{"paper":"/paper/compost-characterizing-and-evaluating","slug":"compost-characterizing-and-evaluating","title":"CoMPosT: Characterizing and Evaluating Caricature in LLM Simulations","date":"2023-10-17","arxiv_id":"2310.11501","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["myracheng/lm_caricature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/entity-matching-using-large-language-models","slug":"entity-matching-using-large-language-models","title":"Entity Matching using Large Language Models","date":"2023-10-17","arxiv_id":"2310.11244","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-llms-for-privilege-escalation","slug":"evaluating-llms-for-privilege-escalation","title":"LLMs as Hackers: Autonomous Linux Privilege Escalation Attacks","date":"2023-10-17","arxiv_id":"2310.11409","n_code_links":1,"syntology":null},{"paper":null,"slug":"experimenting-ai-technologies-for","title":"Experimenting AI Technologies for Disinformation Combat: the IDMO Project","date":"2023-10-17","arxiv_id":"2310.11097","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-prediction-capabilities","title":"Large Language Model Prediction Capabilities: Evidence from a Real-World Forecasting Tournament","date":"2023-10-17","arxiv_id":"2310.13014","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-red-teaming-gender-bias","title":"Learning from Red Teaming: Gender Bias Provocation and Mitigation in Large Language Models","date":"2023-10-17","arxiv_id":"2310.11079","n_code_links":0,"syntology":null},{"paper":"/paper/probing-the-creativity-of-large-language","slug":"probing-the-creativity-of-large-language","title":"Probing the Creativity of Large Language Models: Can models produce divergent semantic association?","date":"2023-10-17","arxiv_id":"2310.11158","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dingnlab/probing_creativity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"what-is-a-good-question-task-oriented-asking","title":"Alexpaca: Learning Factual Clarification Question Generation Without Examples","date":"2023-10-17","arxiv_id":"2310.11571","n_code_links":0,"syntology":null},{"paper":null,"slug":"battle-of-the-large-language-models-dolly-vs","title":"Battle of the Large Language Models: Dolly vs LLaMA vs Vicuna vs Guanaco vs Bard vs ChatGPT -- A Text-to-SQL Parsing Comparison","date":"2023-10-16","arxiv_id":"2310.10190","n_code_links":0,"syntology":null},{"paper":null,"slug":"biomedjourney-counterfactual-biomedical-image","title":"BiomedJourney: Counterfactual Biomedical Image Generation by Instruction-Learning from Multimodal Patient Journeys","date":"2023-10-16","arxiv_id":"2310.10765","n_code_links":0,"syntology":null},{"paper":"/paper/bioplanner-automatic-evaluation-of-llms-on","slug":"bioplanner-automatic-evaluation-of-llms-on","title":"BioPlanner: Automatic Evaluation of LLMs on Protocol Planning in Biology","date":"2023-10-16","arxiv_id":"2310.10632","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bioplanner/bioplanner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/factored-verification-detecting-and-reducing","slug":"factored-verification-detecting-and-reducing","title":"Factored Verification: Detecting and Reducing Hallucination in Summaries of Academic Papers","date":"2023-10-16","arxiv_id":"2310.10627","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompt-packer-deceiving-llms-through","title":"Prompt Packer: Deceiving LLMs through Compositional Instruction with Hidden Attacks","date":"2023-10-16","arxiv_id":"2310.10077","n_code_links":0,"syntology":null},{"paper":"/paper/trigo-benchmarking-formal-mathematical-proof","slug":"trigo-benchmarking-formal-mathematical-proof","title":"TRIGO: Benchmarking Formal Mathematical Proof Reduction for Generative Language Models","date":"2023-10-16","arxiv_id":"2310.10180","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["menik1126/TRIGO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"verbosity-bias-in-preference-labeling-by","title":"Verbosity Bias in Preference Labeling by Large Language Models","date":"2023-10-16","arxiv_id":"2310.10076","n_code_links":0,"syntology":null},{"paper":null,"slug":"diversifying-the-mixture-of-experts","title":"Diversifying the Mixture-of-Experts Representation for Language Models with Orthogonal Optimizer","date":"2023-10-15","arxiv_id":"2310.09762","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-in-context-student","slug":"large-language-models-for-in-context-student","title":"Large Language Models for In-Context Student Modeling: Synthesizing Student's Behavior in Visual Programming","date":"2023-10-15","arxiv_id":"2310.10690","n_code_links":1,"syntology":null},{"paper":"/paper/instruction-tuning-with-human-curriculum","slug":"instruction-tuning-with-human-curriculum","title":"Instruction Tuning with Human Curriculum","date":"2023-10-14","arxiv_id":"2310.09518","n_code_links":1,"syntology":null},{"paper":"/paper/a-systematic-evaluation-of-large-language-1","slug":"a-systematic-evaluation-of-large-language-1","title":"Assessing and Enhancing the Robustness of Large Language Models with Task Structure Variations for Logical Reasoning","date":"2023-10-13","arxiv_id":"2310.09430","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["strong-ai-lab/logical-and-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automated-claim-matching-with-large-language","title":"Automated Claim Matching with Large Language Models: Empowering Fact-Checkers in the Fight Against Misinformation","date":"2023-10-13","arxiv_id":"2310.09223","n_code_links":0,"syntology":null},{"paper":"/paper/dont-add-dont-miss-effective-content","slug":"dont-add-dont-miss-effective-content","title":"Dont Add, dont Miss: Effective Content Preserving Generation from Pre-Selected Text Spans","date":"2023-10-13","arxiv_id":"2310.09017","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lovodkin93/cdr_ctr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/glore-evaluating-logical-reasoning-of-large","slug":"glore-evaluating-logical-reasoning-of-large","title":"GLoRE: Evaluating Logical Reasoning of Large Language Models","date":"2023-10-13","arxiv_id":"2310.09107","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-models-be-financial-analysts-an","title":"Can GPT models be Financial Analysts? An Evaluation of ChatGPT and GPT-4 on mock CFA Exams","date":"2023-10-12","arxiv_id":"2310.08678","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-really-improve-by","title":"Can Large Language Models Really Improve by Self-critiquing Their Own Plans?","date":"2023-10-12","arxiv_id":"2310.08118","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-reward-models-in-rlhf-tuned","slug":"interpreting-reward-models-in-rlhf-tuned","title":"Interpreting Learned Feedback Patterns in Large Language Models","date":"2023-10-12","arxiv_id":"2310.08164","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["apartresearch/interpreting-reward-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-can-replicate-cross","title":"Large language models can replicate cross-cultural differences in personality","date":"2023-10-12","arxiv_id":"2310.10679","n_code_links":0,"syntology":null},{"paper":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","n_code_links":1,"syntology":{"ran":12,"of":12,"n_ran_checked":6,"n_instrument":6,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/prometheus-inducing-fine-grained-evaluation","slug":"prometheus-inducing-fine-grained-evaluation","title":"Prometheus: Inducing Fine-grained Evaluation Capability in Language Models","date":"2023-10-12","arxiv_id":"2310.08491","n_code_links":3,"syntology":{"ran":4,"of":9,"n_ran_checked":3,"n_instrument":1,"unverified":5,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["kaistAI/Prometheus"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/qasina-religious-domain-question-answering","slug":"qasina-religious-domain-question-answering","title":"QASiNa: Religious Domain Question Answering using Sirah Nabawiyah","date":"2023-10-12","arxiv_id":"2310.08102","n_code_links":1,"syntology":null},{"paper":null,"slug":"ziya-vl-bilingual-large-vision-language-model","title":"Ziya-Visual: Bilingual Large Vision-Language Model via Multi-Task Instruction Tuning","date":"2023-10-12","arxiv_id":"2310.08166","n_code_links":0,"syntology":null}],"record_sha256":"e8ef4855fb5b911f02aff507025b6f2950b1371e3a8660aa55d7f20786c2ed5f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}