{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/5","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":29,"rows_per_page":100,"rows":[401,500],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/4","next":"/method/gpt-4/papers/6","papers":[{"paper":null,"slug":"rethinking-emotion-annotations-in-the-era-of","title":"Rethinking Emotion Annotations in the Era of Large Language Models","date":"2024-12-10","arxiv_id":"2412.07906","n_code_links":0,"syntology":null},{"paper":"/paper/towards-automated-cross-domain-exploratory","slug":"towards-automated-cross-domain-exploratory","title":"Towards Automated Cross-domain Exploratory Data Analysis through Large Language Models","date":"2024-12-10","arxiv_id":"2412.07214","n_code_links":2,"syntology":null},{"paper":null,"slug":"anchoring-bias-in-large-language-models-an","title":"Anchoring Bias in Large Language Models: An Experimental Study","date":"2024-12-09","arxiv_id":"2412.06593","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-memorization-and-copyright","title":"Exploring Memorization and Copyright Violation in Frontier LLMs: A Study of the New York Times v. OpenAI 2023 Lawsuit","date":"2024-12-09","arxiv_id":"2412.06370","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-multi-task-learning-for-enhanced","title":"Optimizing Multi-Task Learning for Enhanced Performance in Large Language Models","date":"2024-12-09","arxiv_id":"2412.06249","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-robustness-of-llms-on-crisis","title":"Evaluating Robustness of LLMs on Crisis-Related Microblogs across Events, Information Types, and Linguistic Features","date":"2024-12-08","arxiv_id":"2412.10413","n_code_links":0,"syntology":null},{"paper":"/paper/fully-open-source-moxin-7b-technical-report","slug":"fully-open-source-moxin-7b-technical-report","title":"Fully Open Source Moxin-7B Technical Report","date":"2024-12-08","arxiv_id":"2412.06845","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-correction-explainable-feedback","slug":"learning-to-correction-explainable-feedback","title":"Learning to Correction: Explainable Feedback Generation for Visual Commonsense Reasoning Distractor","date":"2024-12-08","arxiv_id":"2412.07801","n_code_links":1,"syntology":null},{"paper":"/paper/m-3-20m-a-large-scale-multi-modal-molecule","slug":"m-3-20m-a-large-scale-multi-modal-molecule","title":"M$^{3}$-20M: A Large-Scale Multi-Modal Molecule Dataset for AI-driven Drug Design and Discovery","date":"2024-12-08","arxiv_id":"2412.06847","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-use-of-llms-for-sql-equivalence","title":"Can the Rookies Cut the Tough Cookie? Exploring the Use of LLMs for SQL Equivalence Checking","date":"2024-12-07","arxiv_id":"2412.05561","n_code_links":0,"syntology":null},{"paper":null,"slug":"innovative-sentiment-analysis-and-prediction","title":"Innovative Sentiment Analysis and Prediction of Stock Price Using FinBERT, GPT-4 and Logistic Regression: A Data-Driven Approach","date":"2024-12-07","arxiv_id":"2412.06837","n_code_links":0,"syntology":null},{"paper":null,"slug":"shifting-ner-into-high-gear-the-auto-adver","title":"Shifting NER into High Gear: The Auto-AdvER Approach","date":"2024-12-07","arxiv_id":"2412.05655","n_code_links":0,"syntology":null},{"paper":"/paper/towards-learning-to-reason-comparing-llms","slug":"towards-learning-to-reason-comparing-llms","title":"Towards Learning to Reason: Comparing LLMs with Neuro-Symbolic on Arithmetic Relations in Abstract Reasoning","date":"2024-12-07","arxiv_id":"2412.05586","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ibm/raven-large-language-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"100-hallucination-elimination-using-acurai","title":"100% Elimination of Hallucinations on RAGTruth for GPT-4 and GPT-3.5 Turbo","date":"2024-12-06","arxiv_id":"2412.05223","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-frontier-large-language-models-suitable","title":"Are Frontier Large Language Models Suitable for Q&A in Science Centres?","date":"2024-12-06","arxiv_id":"2412.05200","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-llms-for-impression-generation-in","title":"Enhancing LLMs for Impression Generation in Radiology Reports through a Multi-Agent System","date":"2024-12-06","arxiv_id":"2412.06828","n_code_links":0,"syntology":null},{"paper":null,"slug":"addressing-hallucinations-with-rag-and-nmiss","title":"Addressing Hallucinations with RAG and NMISS in Italian Healthcare LLM Chatbots","date":"2024-12-05","arxiv_id":"2412.04235","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-good-is-chatgpt-in-giving-adaptive","title":"How Good is ChatGPT in Giving Adaptive Guidance Using Knowledge Graphs in E-Learning Environments?","date":"2024-12-05","arxiv_id":"2412.03856","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-water-efficiency-dataset-for-african-data","title":"A Water Efficiency Dataset for African Data Centers","date":"2024-12-04","arxiv_id":"2412.03716","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-safety-training-of-llms-generalize-to","title":"Does Safety Training of LLMs Generalize to Semantically Related Natural Prompts?","date":"2024-12-04","arxiv_id":"2412.03235","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-20","title":"Leveraging Large Language Models for Comparative Literature Summarization with Reflective Incremental Mechanisms","date":"2024-12-03","arxiv_id":"2412.02149","n_code_links":0,"syntology":null},{"paper":null,"slug":"patent-cr-a-dataset-for-patent-claim-revision","title":"Patent-CR: A Dataset for Patent Claim Revision","date":"2024-12-03","arxiv_id":"2412.02549","n_code_links":0,"syntology":null},{"paper":"/paper/rare-retrieval-augmented-reasoning","slug":"rare-retrieval-augmented-reasoning","title":"RARE: Retrieval-Augmented Reasoning Enhancement for Large Language Models","date":"2024-12-03","arxiv_id":"2412.02830","n_code_links":1,"syntology":null},{"paper":null,"slug":"automated-extraction-of-acronym-expansion","title":"Automated Extraction of Acronym-Expansion Pairs from Scientific Papers","date":"2024-12-02","arxiv_id":"2412.01093","n_code_links":0,"syntology":null},{"paper":null,"slug":"nyt-connections-a-deceptively-simple-text","title":"NYT-Connections: A Deceptively Simple Text Classification Task that Stumps System-1 Thinkers","date":"2024-12-02","arxiv_id":"2412.01621","n_code_links":0,"syntology":null},{"paper":null,"slug":"pkrd-cot-a-unified-chain-of-thought-prompting","title":"PKRD-CoT: A Unified Chain-of-thought Prompting for Multi-Modal Large Language Models in Autonomous Driving","date":"2024-12-02","arxiv_id":"2412.02025","n_code_links":0,"syntology":null},{"paper":null,"slug":"r-bot-an-llm-based-query-rewrite-system","title":"R-Bot: An LLM-based Query Rewrite System","date":"2024-12-02","arxiv_id":"2412.01661","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-promise-and-peril-of-generative-ai","title":"The Promise and Peril of Generative AI: Evidence from GPT-4 as Sell-Side Analysts","date":"2024-12-02","arxiv_id":"2412.01069","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitive-biases-in-large-language-models-a","title":"Cognitive Biases in Large Language Models: A Survey and Mitigation Experiments","date":"2024-11-30","arxiv_id":"2412.00323","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-domain-specific-post-training-for","title":"On Domain-Specific Post-Training for Multimodal Large Language Models","date":"2024-11-29","arxiv_id":"2411.19930","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-agents-with-weakly-supervised","title":"Training Agents with Weakly Supervised Feedback from Large Language Models","date":"2024-11-29","arxiv_id":"2411.19547","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-lean-dataset-for-international-math","title":"A Lean Dataset for International Math Olympiad: Small Steps towards Writing Math Proofs for Hard Problems","date":"2024-11-28","arxiv_id":"2411.18872","n_code_links":0,"syntology":null},{"paper":null,"slug":"mag-v-a-multi-agent-framework-for-synthetic","title":"MAG-V: A Multi-Agent Framework for Synthetic Data Generation and Verification","date":"2024-11-28","arxiv_id":"2412.04494","n_code_links":0,"syntology":null},{"paper":null,"slug":"matata-a-weak-supervised-mathematical-tool","title":"MATATA: Weakly Supervised End-to-End MAthematical Tool-Augmented Reasoning for Tabular Applications","date":"2024-11-28","arxiv_id":"2411.18915","n_code_links":0,"syntology":null},{"paper":null,"slug":"smartllmsentry-a-comprehensive-llm-based","title":"SmartLLMSentry: A Comprehensive LLM Based Smart Contract Vulnerability Detection Framework","date":"2024-11-28","arxiv_id":"2411.19234","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-example-selection-in-few-shot","title":"The Impact of Example Selection in Few-Shot Prompting on Automated Essay Scoring Using GPT Models","date":"2024-11-28","arxiv_id":"2411.18924","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-pipeline-of-neural-symbolic-integration-to","title":"Dspy-based Neural-Symbolic Pipeline to Enhance Spatial Reasoning in LLMs","date":"2024-11-27","arxiv_id":"2411.18564","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-knowledge-concepts-to-whole-slide","slug":"aligning-knowledge-concepts-to-whole-slide","title":"Aligning Knowledge Concepts to Whole Slide Images for Precise Histopathology Image Analysis","date":"2024-11-27","arxiv_id":"2411.18101","n_code_links":1,"syntology":null},{"paper":"/paper/the-importance-of-visual-modelling-languages","slug":"the-importance-of-visual-modelling-languages","title":"The importance of visual modelling languages in generative software engineering","date":"2024-11-27","arxiv_id":"2411.17976","n_code_links":1,"syntology":null},{"paper":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-artificial-intelligence-predict-clinical","title":"Can artificial intelligence predict clinical trial outcomes?","date":"2024-11-26","arxiv_id":"2411.17595","n_code_links":0,"syntology":null},{"paper":null,"slug":"er2score-llm-based-explainable-and","title":"ER2Score: LLM-based Explainable and Customizable Metric for Assessing Radiology Reports with Reward-Control Loss","date":"2024-11-26","arxiv_id":"2411.17301","n_code_links":0,"syntology":null},{"paper":null,"slug":"give-me-the-code-log-analysis-of-first-year","title":"\"Give me the code\" -- Log Analysis of First-Year CS Students' Interactions With GPT","date":"2024-11-26","arxiv_id":"2411.17855","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-large-language-models-and-topic","slug":"leveraging-large-language-models-and-topic","title":"Leveraging Large Language Models and Topic Modeling for Toxicity Classification","date":"2024-11-26","arxiv_id":"2411.17876","n_code_links":1,"syntology":null},{"paper":"/paper/marvel-40m-multi-level-visual-elaboration-for","slug":"marvel-40m-multi-level-visual-elaboration-for","title":"MARVEL-40M+: Multi-Level Visual Elaboration for High-Fidelity Text-to-3D Content Creation","date":"2024-11-26","arxiv_id":"2411.17945","n_code_links":2,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["SadilKhan/MARVEL-FX3D","huggingface.co/datasets/sankalpsinha77/MARVEL-40M"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"push-the-limit-of-multi-modal-emotion","title":"Push the Limit of Multi-modal Emotion Recognition by Prompting LLMs with Receptive-Field-Aware Attention Weighting","date":"2024-11-26","arxiv_id":"2411.17674","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-ai-grade-your-essays-a-comparative","title":"Can AI grade your essays? A comparative analysis of large language models and teacher ratings in multidimensional essay scoring","date":"2024-11-25","arxiv_id":"2411.16337","n_code_links":0,"syntology":null},{"paper":null,"slug":"catp-llm-empowering-large-language-models-for","title":"CATP-LLM: Empowering Large Language Models for Cost-Aware Tool Planning","date":"2024-11-25","arxiv_id":"2411.16313","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-answer-reliability-through-inter","title":"Enhancing Answer Reliability Through Inter-Model Consensus of Large Language Models","date":"2024-11-25","arxiv_id":"2411.16797","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-factuality-in-long-form-text","title":"Investigating Factuality in Long-Form Text Generation: The Roles of Self-Known and Self-Unknown","date":"2024-11-24","arxiv_id":"2411.15993","n_code_links":0,"syntology":null},{"paper":null,"slug":"don-t-mesh-with-me-generating-constructive","title":"Don't Mesh with Me: Generating Constructive Solid Geometry Instead of Meshes by Fine-Tuning a Code-Generation LLM","date":"2024-11-22","arxiv_id":"2411.15279","n_code_links":0,"syntology":null},{"paper":null,"slug":"purrfessor-a-fine-tuned-multimodal-llava-diet","title":"Purrfessor: A Fine-tuned Multimodal LLaVA Diet Health Chatbot","date":"2024-11-22","arxiv_id":"2411.14925","n_code_links":0,"syntology":null},{"paper":"/paper/scribeagent-towards-specialized-web-agents","slug":"scribeagent-towards-specialized-web-agents","title":"ScribeAgent: Towards Specialized Web Agents Using Production-Scale Workflow Data","date":"2024-11-22","arxiv_id":"2411.15004","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["colonylabs/ScribeAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/benchmarking-gpt-4-against-human-translators","slug":"benchmarking-gpt-4-against-human-translators","title":"Benchmarking GPT-4 against Human Translators: A Comprehensive Evaluation Across Languages, Domains, and Expertise Levels","date":"2024-11-21","arxiv_id":"2411.13775","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["elliottyan/gpt_versus_mt_experts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"explaining-gpt-4-s-schema-of-depression-using","title":"Explaining GPT-4's Schema of Depression Using Machine Behavior Analysis","date":"2024-11-21","arxiv_id":"2411.13800","n_code_links":0,"syntology":null},{"paper":"/paper/gmai-vl-gmai-vl-5-5m-a-large-vision-language","slug":"gmai-vl-gmai-vl-5-5m-a-large-vision-language","title":"GMAI-VL & GMAI-VL-5.5M: A Large Vision-Language Model and A Comprehensive Multimodal Dataset Towards General Medical AI","date":"2024-11-21","arxiv_id":"2411.14522","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":9,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uni-medical/gmai-vl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-from-silly-questions-improves-large","title":"Learning from \"Silly\" Questions Improves Large Language Models, But Only Slightly","date":"2024-11-21","arxiv_id":"2411.14121","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-world-or-predicting-future-a","title":"Understanding World or Predicting Future? A Comprehensive Survey of World Models","date":"2024-11-21","arxiv_id":"2411.14499","n_code_links":0,"syntology":null},{"paper":null,"slug":"bipro-zero-shot-chinese-poem-generation-via","title":"BIPro: Zero-shot Chinese Poem Generation via Block Inverse Prompting Constrained Generation Framework","date":"2024-11-20","arxiv_id":"2411.13237","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-for-climate","title":"Exploring Large Language Models for Climate Forecasting","date":"2024-11-20","arxiv_id":"2411.13724","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impossible-test-a-2024-unsolvable-dataset","title":"The Impossible Test: A 2024 Unsolvable Dataset and A Chance for an AGI Quiz","date":"2024-11-20","arxiv_id":"2411.14486","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-tokenizer-performance-of-large","title":"Evaluating Tokenizer Performance of Large Language Models Across Official Indian Languages","date":"2024-11-19","arxiv_id":"2411.12240","n_code_links":0,"syntology":null},{"paper":null,"slug":"chapter-7-review-of-data-driven-generative-ai","title":"Chapter 7 Review of Data-Driven Generative AI Models for Knowledge Extraction from Scientific Literature in Healthcare","date":"2024-11-18","arxiv_id":"2411.11635","n_code_links":0,"syntology":null},{"paper":"/paper/perfcodegen-improving-performance-of-llm","slug":"perfcodegen-improving-performance-of-llm","title":"PerfCodeGen: Improving Performance of LLM Generated Code with Execution Feedback","date":"2024-11-18","arxiv_id":"2412.03578","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["SalesforceAIResearch/perfcodegen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"popular-llms-amplify-race-and-gender","title":"Popular LLMs Amplify Race and Gender Disparities in Human Mobility","date":"2024-11-18","arxiv_id":"2411.14469","n_code_links":0,"syntology":null},{"paper":null,"slug":"intentgpt-few-shot-intent-discovery-with","title":"IntentGPT: Few-shot Intent Discovery with Large Language Models","date":"2024-11-16","arxiv_id":"2411.10670","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-prompt-formatting-have-any-impact-on-llm","title":"Does Prompt Formatting Have Any Impact on LLM Performance?","date":"2024-11-15","arxiv_id":"2411.10541","n_code_links":0,"syntology":null},{"paper":null,"slug":"lora-litee-a-computationally-efficient","title":"LoRA-LiteE: A Computationally Efficient Framework for Chatbot Preference-Tuning","date":"2024-11-15","arxiv_id":"2411.09947","n_code_links":0,"syntology":null},{"paper":null,"slug":"adopting-rag-for-llm-aided-future-vehicle","title":"Adopting RAG for LLM-Aided Future Vehicle Design","date":"2024-11-14","arxiv_id":"2411.09590","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-autograding-large-language-models","title":"Automating Autograding: Large Language Models as Test Suite Generators for Introductory Programming","date":"2024-11-14","arxiv_id":"2411.09261","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-gender-bias-in-large-language-1","title":"Evaluating Gender Bias in Large Language Models","date":"2024-11-14","arxiv_id":"2411.09826","n_code_links":0,"syntology":null},{"paper":"/paper/mm-eval-a-hierarchical-benchmark-for-modern","slug":"mm-eval-a-hierarchical-benchmark-for-modern","title":"MM-Eval: A Hierarchical Benchmark for Modern Mongolian Evaluation in LLMs","date":"2024-11-14","arxiv_id":"2411.09492","n_code_links":1,"syntology":null},{"paper":null,"slug":"continuous-gnn-based-anomaly-detection-on","title":"Continuous GNN-based Anomaly Detection on Edge using Efficient Adaptive Knowledge Graph Learning","date":"2024-11-13","arxiv_id":"2411.09072","n_code_links":0,"syntology":null},{"paper":null,"slug":"budgetmlagent-a-cost-effective-llm-multi","title":"BudgetMLAgent: A Cost-Effective LLM Multi-Agent system for Automating Machine Learning Tasks","date":"2024-11-12","arxiv_id":"2411.07464","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-chatgpt-3-5-efficiency-in-solving","slug":"evaluating-chatgpt-3-5-efficiency-in-solving","title":"Evaluating ChatGPT-3.5 Efficiency in Solving Coding Problems of Different Complexity Levels: An Empirical Analysis","date":"2024-11-12","arxiv_id":"2411.07529","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-grapheme-to-phoneme-conversion","title":"Improving Grapheme-to-Phoneme Conversion through In-Context Knowledge Retrieval with Large Language Models","date":"2024-11-12","arxiv_id":"2411.07563","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-can-self-improve-in","slug":"large-language-models-can-self-improve-in","title":"Large Language Models Can Self-Improve in Long-context Reasoning","date":"2024-11-12","arxiv_id":"2411.08147","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sihengli99/sealong"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/verbosity-neq-veracity-demystify-verbosity","slug":"verbosity-neq-veracity-demystify-verbosity","title":"Verbosity $\\neq$ Veracity: Demystify Verbosity Compensation Behavior of Large Language Models","date":"2024-11-12","arxiv_id":"2411.07858","n_code_links":1,"syntology":null},{"paper":"/paper/storyteller-improving-long-video-description","slug":"storyteller-improving-long-video-description","title":"StoryTeller: Improving Long Video Description through Global Audio-Visual Character Identification","date":"2024-11-11","arxiv_id":"2411.07076","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-s-spatial-intelligence-evaluating-ai-s","title":"AI's Spatial Intelligence: Evaluating AI's Understanding of Spatial Transformations in PSVT:R and Augmented Reality","date":"2024-11-09","arxiv_id":"2411.06269","n_code_links":0,"syntology":null},{"paper":"/paper/using-language-models-to-disambiguate-lexical","slug":"using-language-models-to-disambiguate-lexical","title":"Using Language Models to Disambiguate Lexical Choices in Translation","date":"2024-11-08","arxiv_id":"2411.05781","n_code_links":1,"syntology":null},{"paper":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","n_code_links":1,"syntology":{"ran":14,"of":16,"n_ran_checked":13,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm2clip-powerful-language-model-unlock","slug":"llm2clip-powerful-language-model-unlock","title":"LLM2CLIP: Powerful Language Model Unlocks Richer Visual Representation","date":"2024-11-07","arxiv_id":"2411.04997","n_code_links":1,"syntology":null},{"paper":"/paper/measuring-short-form-factuality-in-large","slug":"measuring-short-form-factuality-in-large","title":"Measuring short-form factuality in large language models","date":"2024-11-07","arxiv_id":"2411.04368","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-recent-large-language","title":"A Comparative Study of Recent Large Language Models on Generating Hospital Discharge Summaries for Lung Cancer Patients","date":"2024-11-06","arxiv_id":"2411.03805","n_code_links":0,"syntology":null},{"paper":"/paper/customized-multiple-clustering-via-multi","slug":"customized-multiple-clustering-via-multi","title":"Customized Multiple Clustering via Multi-Modal Subspace Proxy Learning","date":"2024-11-06","arxiv_id":"2411.03978","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":0,"n_instrument":3,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alexander-yao/multi-sub"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"diversity-helps-jailbreak-large-language","title":"Diversity Helps Jailbreak Large Language Models","date":"2024-11-06","arxiv_id":"2411.04223","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-medprompt-to-o1-exploration-of-run-time","title":"From Medprompt to o1: Exploration of Run-Time Strategies for Medical Challenge Problems and Beyond","date":"2024-11-06","arxiv_id":"2411.03590","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-quantification-for-clinical","title":"Uncertainty Quantification for Clinical Outcome Predictions with (Large) Language Models","date":"2024-11-05","arxiv_id":"2411.03497","n_code_links":0,"syntology":null},{"paper":null,"slug":"user-centric-semantic-communications","title":"Receiver-Centric Generative Semantic Communications","date":"2024-11-05","arxiv_id":"2411.03127","n_code_links":0,"syntology":null},{"paper":null,"slug":"veritas-a-unified-approach-to-reliability","title":"VERITAS: A Unified Approach to Reliability Evaluation","date":"2024-11-05","arxiv_id":"2411.03300","n_code_links":0,"syntology":null},{"paper":null,"slug":"disrupting-test-development-with-ai","title":"Disrupting Test Development with AI Assistants","date":"2024-11-04","arxiv_id":"2411.02328","n_code_links":0,"syntology":null},{"paper":"/paper/seq-vcr-preventing-collapse-in-intermediate","slug":"seq-vcr-preventing-collapse-in-intermediate","title":"Seq-VCR: Preventing Collapse in Intermediate Transformer Representations for Enhanced Reasoning","date":"2024-11-04","arxiv_id":"2411.02344","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-leveraging-news-media-to-support","title":"Towards Leveraging News Media to Support Impact Assessment of AI Technologies","date":"2024-11-04","arxiv_id":"2411.02536","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-deep-dive-into-large-language-model-code","title":"A Deep Dive Into Large Language Model Code Generation Mistakes: What and Why?","date":"2024-11-03","arxiv_id":"2411.01414","n_code_links":0,"syntology":null},{"paper":null,"slug":"high-performance-automated-abstract-screening","title":"High-performance automated abstract screening with large language model ensembles","date":"2024-11-03","arxiv_id":"2411.02451","n_code_links":0,"syntology":null},{"paper":null,"slug":"integration-of-large-vision-language-models","title":"Integration of Large Vision Language Models for Efficient Post-disaster Damage Assessment and Reporting","date":"2024-11-03","arxiv_id":"2411.01511","n_code_links":0,"syntology":null},{"paper":null,"slug":"uniguard-towards-universal-safety-guardrails","title":"UniGuard: Towards Universal Safety Guardrails for Jailbreak Attacks on Multimodal Large Language Models","date":"2024-11-03","arxiv_id":"2411.01703","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoning-limitations-of-multimodal-large","title":"Reasoning Limitations of Multimodal Large Language Models. A case study of Bongard Problems","date":"2024-11-02","arxiv_id":"2411.01173","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-impact-of-lab-test-results-on","title":"Evaluating the Impact of Lab Test Results on Large Language Models Generated Differential Diagnoses from Clinical Case Vignettes","date":"2024-11-01","arxiv_id":"2411.02523","n_code_links":0,"syntology":null}],"record_sha256":"024f38ed01e16aa03ae6fa080628ad7eae2c0b4eff56e5d0d1c5b61809b58088","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}