{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/21","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":21,"pages_in_order":29,"rows_per_page":100,"rows":[2001,2100],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/20","next":"/method/gpt-4/papers/22","papers":[{"paper":null,"slug":"large-language-model-llm-bias-index-llmbi","title":"Large Language Model (LLM) Bias Index -- LLMBI","date":"2023-12-22","arxiv_id":"2312.14769","n_code_links":0,"syntology":null},{"paper":null,"slug":"mmgpl-multimodal-medical-data-analysis-with","title":"MMGPL: Multimodal Medical Data Analysis with Graph Prompt Learning","date":"2023-12-22","arxiv_id":"2312.14574","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-unified-multimodal-reasoning","slug":"towards-a-unified-multimodal-reasoning","title":"Towards a Unified Multimodal Reasoning Framework","date":"2023-12-22","arxiv_id":"2312.15021","n_code_links":1,"syntology":null},{"paper":null,"slug":"voila-a-aligning-vision-language-models-with","title":"Voila-A: Aligning Vision-Language Models with User's Gaze Attention","date":"2023-12-22","arxiv_id":"2401.09454","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-novel-gpt-4-apis","slug":"exploiting-novel-gpt-4-apis","title":"Exploiting Novel GPT-4 APIs","date":"2023-12-21","arxiv_id":"2312.14302","n_code_links":1,"syntology":null},{"paper":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","n_code_links":2,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"shai-a-large-language-model-for-asset","title":"Shai: A large language model for asset management","date":"2023-12-21","arxiv_id":"2312.14203","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-gpt-4-prompts-to-determine-whether","title":"Preparing to Integrate Generative Pretrained Transformer Series 4 models into Genetic Variant Assessment Workflows: Assessing Performance, Drift, and Nondeterminism Characteristics Relative to Classifying Functional Evidence in Literature","date":"2023-12-21","arxiv_id":"2312.13521","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-and-analyzing-in-context","title":"Benchmarking and Analyzing In-context Learning, Fine-tuning and Supervised Learning for Biomedical Knowledge Curation: a focused study on chemical entities of biological interest","date":"2023-12-20","arxiv_id":"2312.12989","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-down-to-scale-up-a-cost-benefit","slug":"scaling-down-to-scale-up-a-cost-benefit","title":"Scaling Down to Scale Up: A Cost-Benefit Analysis of Replacing OpenAI's LLM with Open Source SLMs in Production","date":"2023-12-20","arxiv_id":"2312.14972","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-in-medical-term","title":"Large Language Models in Medical Term Classification and Unexpected Misalignment Between Response and Reasoning","date":"2023-12-19","arxiv_id":"2312.14184","n_code_links":0,"syntology":null},{"paper":"/paper/llm-ark-knowledge-graph-reasoning-using-large","slug":"llm-ark-knowledge-graph-reasoning-using-large","title":"Evaluating and Enhancing Large Language Models for Conversational Reasoning on Knowledge Graphs","date":"2023-12-18","arxiv_id":"2312.11282","n_code_links":1,"syntology":null},{"paper":"/paper/mac-sql-multi-agent-collaboration-for-text-to","slug":"mac-sql-multi-agent-collaboration-for-text-to","title":"MAC-SQL: A Multi-Agent Collaborative Framework for Text-to-SQL","date":"2023-12-18","arxiv_id":"2312.11242","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wbbeyourself/mac-sql"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/nomiracl-knowing-when-you-don-t-know-for","slug":"nomiracl-knowing-when-you-don-t-know-for","title":"\"Knowing When You Don't Know\": A Multilingual Relevance Assessment Dataset for Robust Retrieval-Augmented Generation","date":"2023-12-18","arxiv_id":"2312.11361","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["project-miracl/nomiracl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-evaluation-of-gpt-4v-and-gemini-in-online","title":"An Evaluation of GPT-4V and Gemini in Online VQA","date":"2023-12-17","arxiv_id":"2312.10637","n_code_links":0,"syntology":null},{"paper":null,"slug":"ceir-concept-based-explainable-image","title":"CEIR: Concept-based Explainable Image Representation Learning","date":"2023-12-17","arxiv_id":"2312.10747","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-analysis-of-large-language","title":"A Comparative Analysis of Large Language Models for Code Documentation Generation","date":"2023-12-16","arxiv_id":"2312.10349","n_code_links":0,"syntology":null},{"paper":null,"slug":"deepart-a-benchmark-to-advance-fidelity","title":"DeepArt: A Benchmark to Advance Fidelity Research in AI-Generated Content","date":"2023-12-16","arxiv_id":"2312.10407","n_code_links":0,"syntology":null},{"paper":"/paper/recprompt-a-prompt-tuning-framework-for-news","slug":"recprompt-a-prompt-tuning-framework-for-news","title":"RecPrompt: A Self-tuning Prompting Framework for News Recommendation Using Large Language Models","date":"2023-12-16","arxiv_id":"2312.10463","n_code_links":1,"syntology":null},{"paper":"/paper/binary-code-summarization-benchmarking","slug":"binary-code-summarization-benchmarking","title":"Binary Code Summarization: Benchmarking ChatGPT/GPT-4 and Other Large Language Models","date":"2023-12-15","arxiv_id":"2312.09601","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-large-language-models-for-matching","title":"Distilling Large Language Models for Matching Patients to Clinical Trials","date":"2023-12-15","arxiv_id":"2312.09958","n_code_links":0,"syntology":null},{"paper":null,"slug":"integrating-ai-and-learning-analytics-for","title":"Integrating AI and Learning Analytics for Data-Driven Pedagogical Decisions and Personalized Interventions in Education","date":"2023-12-15","arxiv_id":"2312.09548","n_code_links":0,"syntology":null},{"paper":"/paper/openmedcalc-augmentation-of-chatgpt-with","slug":"openmedcalc-augmentation-of-chatgpt-with","title":"OpenMedCalc: Augmentation of ChatGPT with Clinician-Informed Tools Improves Performance on Medical Calculation Tasks","date":"2023-12-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/heterogeneous-graph-neural-architecture","slug":"heterogeneous-graph-neural-architecture","title":"Heterogeneous Graph Neural Architecture Search with GPT-4","date":"2023-12-14","arxiv_id":"2312.08680","n_code_links":1,"syntology":null},{"paper":"/paper/holodeck-language-guided-generation-of-3d","slug":"holodeck-language-guided-generation-of-3d","title":"Holodeck: Language Guided Generation of 3D Embodied AI Environments","date":"2023-12-14","arxiv_id":"2312.09067","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/Holodeck"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/modeling-complex-mathematical-reasoning-via","slug":"modeling-complex-mathematical-reasoning-via","title":"Modeling Complex Mathematical Reasoning via Large Language Model based MathAgent","date":"2023-12-14","arxiv_id":"2312.08926","n_code_links":1,"syntology":null},{"paper":null,"slug":"weak-to-strong-generalization-eliciting","title":"Weak-to-Strong Generalization: Eliciting Strong Capabilities With Weak Supervision","date":"2023-12-14","arxiv_id":"2312.09390","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-gpt4-v-on-structured-reasoning","title":"Assessing GPT4-V on Structured Reasoning Tasks","date":"2023-12-13","arxiv_id":"2312.11524","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-english-evaluating-llms-for-arabic","title":"Beyond English: Evaluating LLMs for Arabic Grammatical Error Correction","date":"2023-12-13","arxiv_id":"2312.08400","n_code_links":0,"syntology":null},{"paper":"/paper/chat-3d-v2-bridging-3d-scene-and-large","slug":"chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","arxiv_id":"2312.08168","n_code_links":2,"syntology":{"ran":11,"of":11,"n_ran_checked":6,"n_instrument":5,"unverified":0,"pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chat-3d/chat-3d-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"coie-chain-of-instruct-editing-for-multi","title":"CoIE: Chain-of-Instruct Editing for Multi-Attribute Face Manipulation","date":"2023-12-13","arxiv_id":"2312.07879","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-throughput-biomedical-relation","title":"High-throughput Biomedical Relation Extraction for Semi-Structured Web Articles Empowered by Large Language Models","date":"2023-12-13","arxiv_id":"2312.08274","n_code_links":0,"syntology":null},{"paper":null,"slug":"native-language-identification-with-large","title":"Native Language Identification with Large Language Models","date":"2023-12-13","arxiv_id":"2312.07819","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-engineering-assisted-malware-dynamic","slug":"prompt-engineering-assisted-malware-dynamic","title":"Prompt Engineering-assisted Malware Dynamic Analysis Using GPT-4","date":"2023-12-13","arxiv_id":"2312.08317","n_code_links":1,"syntology":null},{"paper":"/paper/ai-control-improving-safety-despite","slug":"ai-control-improving-safety-despite","title":"AI Control: Improving Safety Despite Intentional Subversion","date":"2023-12-12","arxiv_id":"2312.06942","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rgreenblatt/control-evaluations"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/context-matter-data-efficient-augmentation-of","slug":"context-matter-data-efficient-augmentation-of","title":"Context Matters: Data-Efficient Augmentation of Large Language Models for Scientific Applications","date":"2023-12-12","arxiv_id":"2312.07069","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-to-facilitate","title":"Exploring Large Language Models to Facilitate Variable Autonomy for Human-Robot Teaming","date":"2023-12-12","arxiv_id":"2312.07214","n_code_links":0,"syntology":null},{"paper":"/paper/large-foundation-models-for-power-systems","slug":"large-foundation-models-for-power-systems","title":"Large Foundation Models for Power Systems","date":"2023-12-12","arxiv_id":"2312.07044","n_code_links":1,"syntology":null},{"paper":null,"slug":"llmeval-a-preliminary-study-on-how-to","title":"LLMEval: A Preliminary Study on How to Evaluate Large Language Models","date":"2023-12-12","arxiv_id":"2312.07398","n_code_links":0,"syntology":null},{"paper":"/paper/safety-alignment-in-nlp-tasks-weakly-aligned","slug":"safety-alignment-in-nlp-tasks-weakly-aligned","title":"Safety Alignment in NLP Tasks: Weakly Aligned Summarization as an In-Context Attack","date":"2023-12-12","arxiv_id":"2312.06924","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fyyfu/safetyalignnlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"audio-visual-llm-for-video-understanding","title":"Audio-Visual LLM for Video Understanding","date":"2023-12-11","arxiv_id":"2312.06720","n_code_links":0,"syntology":null},{"paper":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":6,"n_instrument":3,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gptbias-a-comprehensive-framework-for","title":"GPTBIAS: A Comprehensive Framework for Evaluating Bias in Large Language Models","date":"2023-12-11","arxiv_id":"2312.06315","n_code_links":0,"syntology":null},{"paper":"/paper/gta-gated-toxicity-avoidance-for-lm","slug":"gta-gated-toxicity-avoidance-for-lm","title":"GTA: Gated Toxicity Avoidance for LM Performance Preservation","date":"2023-12-11","arxiv_id":"2312.06122","n_code_links":1,"syntology":null},{"paper":null,"slug":"interactive-planning-using-large-language","title":"Interactive Planning Using Large Language Models for Partially Observable Robotics Tasks","date":"2023-12-11","arxiv_id":"2312.06876","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowgpt-black-box-knowledge-injection-for","title":"KnowGPT: Knowledge Graph based Prompting for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06185","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-tuning-for-retrieval-augmented","title":"Context Tuning for Retrieval Augmented Generation","date":"2023-12-09","arxiv_id":"2312.05708","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-and-safety-case-generation-an","title":"GPT-4 and Safety Case Generation: An Exploratory Analysis","date":"2023-12-09","arxiv_id":"2312.05696","n_code_links":0,"syntology":null},{"paper":"/paper/sim-gpt-text-similarity-via-gpt-annotated","slug":"sim-gpt-text-similarity-via-gpt-annotated","title":"Sim-GPT: Text Similarity via GPT Annotated Data","date":"2023-12-09","arxiv_id":"2312.05603","n_code_links":1,"syntology":null},{"paper":"/paper/apollo-s-oracle-retrieval-augmented-reasoning","slug":"apollo-s-oracle-retrieval-augmented-reasoning","title":"Learning to Break: Knowledge-Enhanced Reasoning in Multi-Agent Debate System","date":"2023-12-08","arxiv_id":"2312.04854","n_code_links":2,"syntology":null},{"paper":null,"slug":"exploring-the-limits-of-chatgpt-in-software","title":"Exploring the Limits of ChatGPT in Software Security Applications","date":"2023-12-08","arxiv_id":"2312.05275","n_code_links":0,"syntology":null},{"paper":"/paper/kwaiagents-generalized-information-seeking","slug":"kwaiagents-generalized-information-seeking","title":"KwaiAgents: Generalized Information-seeking Agent System with Large Language Models","date":"2023-12-08","arxiv_id":"2312.04889","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwaikeg/kwaiagents"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/pixlore-a-dataset-driven-approach-to-rich","slug":"pixlore-a-dataset-driven-approach-to-rich","title":"PixLore: A Dataset-driven Approach to Rich Image Captioning","date":"2023-12-08","arxiv_id":"2312.05349","n_code_links":1,"syntology":null},{"paper":"/paper/cost-effective-in-context-learning-for-entity","slug":"cost-effective-in-context-learning-for-entity","title":"Cost-Effective In-Context Learning for Entity Resolution: A Design Space Exploration","date":"2023-12-07","arxiv_id":"2312.03987","n_code_links":1,"syntology":null},{"paper":"/paper/fortify-the-shortest-stave-in-attention","slug":"fortify-the-shortest-stave-in-attention","title":"Fortify the Shortest Stave in Attention: Enhancing Context Awareness of Large Language Models for Effective Tool Use","date":"2023-12-07","arxiv_id":"2312.04455","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fiorina1212/attention-buckets"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-4v-with-emotion-a-zero-shot-benchmark-for","slug":"gpt-4v-with-emotion-a-zero-shot-benchmark-for","title":"GPT-4V with Emotion: A Zero-shot Benchmark for Generalized Emotion Recognition","date":"2023-12-07","arxiv_id":"2312.04293","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zeroqiaoba/gpt4v-emotion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt4sgg-synthesizing-scene-graphs-from","slug":"gpt4sgg-synthesizing-scene-graphs-from","title":"GPT4SGG: Synthesizing Scene Graphs from Holistic and Region-specific Narratives","date":"2023-12-07","arxiv_id":"2312.04314","n_code_links":1,"syntology":null},{"paper":"/paper/lampilot-an-open-benchmark-dataset-for","slug":"lampilot-an-open-benchmark-dataset-for","title":"LaMPilot: An Open Benchmark Dataset for Autonomous Driving with Language Model Programs","date":"2023-12-07","arxiv_id":"2312.04372","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-sarcasm-detection-with-openai-gpt-based","title":"On Sarcasm Detection with OpenAI GPT-based Models","date":"2023-12-07","arxiv_id":"2312.04642","n_code_links":0,"syntology":null},{"paper":"/paper/quilt-llava-visual-instruction-tuning-by","slug":"quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","arxiv_id":"2312.04746","n_code_links":2,"syntology":{"ran":11,"of":12,"n_ran_checked":8,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/gpt-4-enhanced-multimodal-grounding-for","slug":"gpt-4-enhanced-multimodal-grounding-for","title":"GPT-4 Enhanced Multimodal Grounding for Autonomous Driving: Leveraging Cross-Modal Attention with Large Language Models","date":"2023-12-06","arxiv_id":"2312.03543","n_code_links":1,"syntology":null},{"paper":null,"slug":"xaiqa-explainer-based-data-augmentation-for","title":"XAIQA: Explainer-Based Data Augmentation for Extractive Question Answering","date":"2023-12-06","arxiv_id":"2312.03567","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-ai-generated-gpt-4-and","title":"A Comparative Study of AI-Generated (GPT-4) and Human-crafted MCQs in Programming Education","date":"2023-12-05","arxiv_id":"2312.03173","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-vs-human-for-scientific-reviews-a-dual","title":"GPT vs Human for Scientific Reviews: A Dual Source Review on Applications of ChatGPT in Science","date":"2023-12-05","arxiv_id":"2312.03769","n_code_links":0,"syntology":null},{"paper":"/paper/let-the-llms-talk-simulating-human-to-human","slug":"let-the-llms-talk-simulating-human-to-human","title":"Let the LLMs Talk: Simulating Human-to-Human Conversational QA via Zero-Shot LLM-to-LLM Interactions","date":"2023-12-05","arxiv_id":"2312.02913","n_code_links":1,"syntology":null},{"paper":null,"slug":"rank-without-gpt-building-gpt-independent","title":"Rank-without-GPT: Building GPT-Independent Listwise Rerankers on Open-Source Large Language Models","date":"2023-12-05","arxiv_id":"2312.02969","n_code_links":0,"syntology":null},{"paper":"/paper/rankzephyr-effective-and-robust-zero-shot","slug":"rankzephyr-effective-and-robust-zero-shot","title":"RankZephyr: Effective and Robust Zero-Shot Listwise Reranking is a Breeze!","date":"2023-12-05","arxiv_id":"2312.02724","n_code_links":2,"syntology":null},{"paper":null,"slug":"competition-level-problems-are-effective","title":"Competition-Level Problems are Effective LLM Evaluators","date":"2023-12-04","arxiv_id":"2312.02143","n_code_links":0,"syntology":null},{"paper":null,"slug":"explore-select-derive-and-recall-augmenting","title":"Explore, Select, Derive, and Recall: Augmenting LLM with Human-like Memory for Mobile Task Automation","date":"2023-12-04","arxiv_id":"2312.03003","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-language-models-for-context","title":"Fine-Tuning Language Models for Context-Specific SQL Query Generation","date":"2023-12-04","arxiv_id":"2312.02251","n_code_links":0,"syntology":null},{"paper":"/paper/instructta-instruction-tuned-targeted-attack","slug":"instructta-instruction-tuned-targeted-attack","title":"InstructTA: Instruction-Tuned Targeted Attack for Large Vision-Language Models","date":"2023-12-04","arxiv_id":"2312.01886","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":6,"n_instrument":2,"unverified":4,"pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["xunguangwang/instructta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"retrieval-augmented-multi-modal-chain-of","title":"Retrieval-augmented Multi-modal Chain-of-Thoughts Reasoning for Large Language Models","date":"2023-12-04","arxiv_id":"2312.01714","n_code_links":0,"syntology":null},{"paper":"/paper/tree-of-attacks-jailbreaking-black-box-llms","slug":"tree-of-attacks-jailbreaking-black-box-llms","title":"Tree of Attacks: Jailbreaking Black-Box LLMs Automatically","date":"2023-12-04","arxiv_id":"2312.02119","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ricommunity/tap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/d-bot-database-diagnosis-system-using-large","slug":"d-bot-database-diagnosis-system-using-large","title":"D-Bot: Database Diagnosis System using Large Language Models","date":"2023-12-03","arxiv_id":"2312.01454","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tsinghuadatabasegroup/db-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"axiomatic-preference-modeling-for-longform","title":"Axiomatic Preference Modeling for Longform Question Answering","date":"2023-12-02","arxiv_id":"2312.02206","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-voices-to-validity-leveraging-large","title":"From Voices to Validity: Leveraging Large Language Models (LLMs) for Textual Analysis of Policy Stakeholder Interviews","date":"2023-12-02","arxiv_id":"2312.01202","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bayesian-approach-for-prompt-optimization","title":"A Bayesian approach for prompt optimization in pre-trained language models","date":"2023-12-01","arxiv_id":"2312.00471","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-large-language-models-and-chain-of","title":"Applying Large Language Models and Chain-of-Thought for Automatic Scoring","date":"2023-11-30","arxiv_id":"2312.03748","n_code_links":0,"syntology":null},{"paper":"/paper/critiquellm-scaling-llm-as-critic-for","slug":"critiquellm-scaling-llm-as-critic-for","title":"CritiqueLLM: Towards an Informative Critique Generation Model for Evaluation of Large Language Model Generation","date":"2023-11-30","arxiv_id":"2311.18702","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thu-coai/critiquellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/unnatural-error-correction-gpt-4-can-almost","slug":"unnatural-error-correction-gpt-4-can-almost","title":"Unnatural Error Correction: GPT-4 Can Almost Perfectly Handle Unnatural Scrambled Text","date":"2023-11-30","arxiv_id":"2311.18805","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ccqq77/unnatural-error-correction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"aviationgpt-a-large-language-model-for-the","title":"AviationGPT: A Large Language Model for the Aviation Domain","date":"2023-11-29","arxiv_id":"2311.17686","n_code_links":0,"syntology":null},{"paper":"/paper/biomedical-knowledge-graph-enhanced-prompt","slug":"biomedical-knowledge-graph-enhanced-prompt","title":"Biomedical knowledge graph-optimized prompt generation for large language models","date":"2023-11-29","arxiv_id":"2311.17330","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["BaranziniLab/KG_RAG"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/chatillusion-efficient-aligning-interleaved","slug":"chatillusion-efficient-aligning-interleaved","title":"M$^{2}$Chat: Empowering VLM for Multimodal LLM Interleaved Text-Image Generation","date":"2023-11-29","arxiv_id":"2311.17963","n_code_links":1,"syntology":null},{"paper":null,"slug":"grounding-foundation-models-through-federated","title":"Grounding Foundation Models through Federated Transfer Learning: A General Framework","date":"2023-11-29","arxiv_id":"2311.17431","n_code_links":0,"syntology":null},{"paper":null,"slug":"mm-narrator-narrating-long-form-videos-with","title":"MM-Narrator: Narrating Long-form Videos with Multimodal In-Context Learning","date":"2023-11-29","arxiv_id":"2311.17435","n_code_links":0,"syntology":null},{"paper":"/paper/timebench-a-comprehensive-evaluation-of","slug":"timebench-a-comprehensive-evaluation-of","title":"TimeBench: A Comprehensive Evaluation of Temporal Reasoning Abilities in Large Language Models","date":"2023-11-29","arxiv_id":"2311.17667","n_code_links":1,"syntology":null},{"paper":"/paper/can-generalist-foundation-models-outcompete","slug":"can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","arxiv_id":"2311.16452","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"cole-a-hierarchical-generation-framework-for","title":"COLE: A Hierarchical Generation Framework for Multi-Layered and Editable Graphic Design","date":"2023-11-28","arxiv_id":"2311.16974","n_code_links":0,"syntology":null},{"paper":null,"slug":"general-purpose-vs-domain-adapted-large","title":"General-Purpose vs. Domain-Adapted Large Language Models for Extraction of Structured Data from Chest Radiology Reports","date":"2023-11-28","arxiv_id":"2311.17213","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-political-texts-with-chatgpt","title":"Positioning Political Texts with Large Language Models by Asking and Averaging","date":"2023-11-28","arxiv_id":"2311.16639","n_code_links":0,"syntology":null},{"paper":"/paper/the-falcon-series-of-open-language-models","slug":"the-falcon-series-of-open-language-models","title":"The Falcon Series of Open Language Models","date":"2023-11-28","arxiv_id":"2311.16867","n_code_links":0,"syntology":null},{"paper":"/paper/can-vision-language-models-think-from-a-first","slug":"can-vision-language-models-think-from-a-first","title":"EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models","date":"2023-11-27","arxiv_id":"2311.15596","n_code_links":1,"syntology":null},{"paper":null,"slug":"chartllama-a-multimodal-llm-for-chart","title":"ChartLlama: A Multimodal LLM for Chart Understanding and Generation","date":"2023-11-27","arxiv_id":"2311.16483","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-logic-errors-a-comparative-study-on","title":"Decoding Logic Errors: A Comparative Study on Bug Detection by Students and Large Language Models","date":"2023-11-27","arxiv_id":"2311.16017","n_code_links":0,"syntology":null},{"paper":"/paper/gpt4vis-what-can-gpt-4-do-for-zero-shot","slug":"gpt4vis-what-can-gpt-4-do-for-zero-shot","title":"GPT4Vis: What Can GPT-4 Do for Zero-shot Visual Recognition?","date":"2023-11-27","arxiv_id":"2311.15732","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whwu95/GPT4Vis"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"instruct2attack-language-guided-semantic","title":"Instruct2Attack: Language-Guided Semantic Adversarial Attacks","date":"2023-11-27","arxiv_id":"2311.15551","n_code_links":0,"syntology":null},{"paper":"/paper/meditron-70b-scaling-medical-pretraining-for","slug":"meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","arxiv_id":"2311.16079","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":9,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["epfllm/meditron"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-vision-enhancing-llms-empowering","title":"Towards Vision Enhancing LLMs: Empowering Multimodal Knowledge Storage and Sharing in LLMs","date":"2023-11-27","arxiv_id":"2311.15759","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-ai-chatbots-performance-in","title":"Comparative Analysis of ChatGPT, GPT-4, and Microsoft Bing Chatbots for GRE Test","date":"2023-11-26","arxiv_id":"2312.03719","n_code_links":0,"syntology":null},{"paper":"/paper/autoeval-video-an-automatic-benchmark-for","slug":"autoeval-video-an-automatic-benchmark-for","title":"AutoEval-Video: An Automatic Benchmark for Assessing Large Vision Language Models in Open-Ended Video Question Answering","date":"2023-11-25","arxiv_id":"2311.14906","n_code_links":1,"syntology":null}],"record_sha256":"4d2f745f3061fafbe1098cd5fc607e8965353486d5cf530e3687d6a0fef0f359","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}