{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/7","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":12,"rows_per_page":100,"rows":[601,700],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/6","next":"/task/multiple-choice/papers/8","papers":[{"url":null,"slug":"a-semantic-parsing-algorithm-to-solve-linear","title":"A Semantic Parsing Algorithm to Solve Linear Ordering Problems","date":"2025-02-12","arxiv_id":"2502.08415","repositories_listed":0,"syntology":null},{"url":null,"slug":"break-the-checkbox-challenging-closed-style","title":"Break the Checkbox: Challenging Closed-Style Evaluations of Cultural Alignment in LLMs","date":"2025-02-12","arxiv_id":"2502.08045","repositories_listed":0,"syntology":null},{"url":null,"slug":"sb-bench-stereotype-bias-benchmark-for-large","title":"SB-Bench: Stereotype Bias Benchmark for Large Multimodal Models","date":"2025-02-12","arxiv_id":"2502.08779","repositories_listed":0,"syntology":null},{"url":null,"slug":"percul-a-story-driven-cultural-evaluation-of","title":"PerCul: A Story-Driven Cultural Evaluation of LLMs in Persian","date":"2025-02-11","arxiv_id":"2502.07459","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenization-standards-for-linguistic","title":"Tokenization Standards for Linguistic Integrity: Turkish as a Benchmark","date":"2025-02-10","arxiv_id":"2502.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"llms-to-support-a-domain-specific-knowledge","title":"LLMs to Support a Domain Specific Knowledge Assistant","date":"2025-02-06","arxiv_id":"2502.04095","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-order-effect-investigating-prompt","title":"The Order Effect: Investigating Prompt Sensitivity to Input Order in LLMs","date":"2025-02-06","arxiv_id":"2502.04134","repositories_listed":0,"syntology":null},{"url":null,"slug":"evalita-llm-benchmarking-large-language","title":"Evalita-LLM: Benchmarking Large Language Models on Italian","date":"2025-02-04","arxiv_id":"2502.02289","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-use-of-artificial-intelligence-tools-in","title":"The Use of Artificial Intelligence Tools in Assessing Content Validity: A Comparative Study with Human Experts","date":"2025-02-03","arxiv_id":"2503.15525","repositories_listed":0,"syntology":null},{"url":null,"slug":"coddllm-empowering-large-language-models-for","title":"CoddLLM: Empowering Large Language Models for Data Analytics","date":"2025-02-01","arxiv_id":"2502.00329","repositories_listed":0,"syntology":null},{"url":null,"slug":"innerthoughts-disentangling-representations","title":"InnerThoughts: Disentangling Representations and Predictions in Large Language Models","date":"2025-01-29","arxiv_id":"2501.17994","repositories_listed":0,"syntology":null},{"url":null,"slug":"attribution-analysis-of-legal-language-as","title":"Attribution analysis of legal language as used by LLM","date":"2025-01-28","arxiv_id":"2501.17330","repositories_listed":0,"syntology":null},{"url":null,"slug":"inferring-from-logits-exploring-best","title":"Inferring from Logits: Exploring Best Practices for Decoding-Free Generative Candidate Selection","date":"2025-01-28","arxiv_id":"2501.17338","repositories_listed":0,"syntology":null},{"url":null,"slug":"town-hall-debate-prompting-enhancing-logical","title":"Town Hall Debate Prompting: Enhancing Logical Reasoning in LLMs through Multi-Persona Interaction","date":"2025-01-28","arxiv_id":"2502.15725","repositories_listed":0,"syntology":null},{"url":null,"slug":"options-aware-dense-retrieval-for-multiple","title":"Options-Aware Dense Retrieval for Multiple-Choice query Answering","date":"2025-01-27","arxiv_id":"2501.16111","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardml-a-benchmark-for-evaluating-data","title":"HardML: A Benchmark For Evaluating Data Science And Machine Learning knowledge and reasoning in AI","date":"2025-01-26","arxiv_id":"2501.15627","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-evaluation-based-on-aerospace","title":"LLM Evaluation Based on Aerospace Manufacturing Expertise: Automated Generation and Multi-Model Question Answering","date":"2025-01-25","arxiv_id":"2501.17183","repositories_listed":0,"syntology":null},{"url":null,"slug":"longreason-a-synthetic-long-context-reasoning","title":"LongReason: A Synthetic Long-Context Reasoning Benchmark via Context Expansion","date":"2025-01-25","arxiv_id":"2501.15089","repositories_listed":0,"syntology":null},{"url":null,"slug":"humanity-s-last-exam","title":"Humanity's Last Exam","date":"2025-01-24","arxiv_id":"2501.14249","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-evaluation-a-critical-measure-in-driving","title":"Auto-Evaluation: A Critical Measure in Driving Improvements in Quality and Safety of AI-Generated Lesson Resources","date":"2025-01-23","arxiv_id":"2502.10410","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-reasoning-capacity-of-ai-models-and","title":"On the Reasoning Capacity of AI Models and How to Quantify It","date":"2025-01-23","arxiv_id":"2501.13833","repositories_listed":0,"syntology":null},{"url":null,"slug":"people-reduce-workers-compensation-for-using","title":"The AI Penalization Effect: People Reduce Compensation for Workers Who Use AI","date":"2025-01-22","arxiv_id":"2501.13228","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-plausible-distractors-for-multiple","title":"Generating Plausible Distractors for Multiple-Choice Questions via Student Choice Prediction","date":"2025-01-21","arxiv_id":"2501.13125","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-multimodal-llms-do-visual-temporal","title":"Can Multimodal LLMs do Visual Temporal Understanding and Reasoning? The answer is No!","date":"2025-01-18","arxiv_id":"2501.10674","repositories_listed":0,"syntology":null},{"url":null,"slug":"empowering-large-language-models-in-wireless","title":"Empowering Large Language Models in Wireless Communication: A Novel Dataset and Fine-Tuning Framework","date":"2025-01-16","arxiv_id":"2501.09631","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-choice-questions-reasoning-makes","title":"Multiple Choice Questions: Reasoning Makes Large Language Models (LLMs) More Self-Confident Even When They Are Wrong","date":"2025-01-16","arxiv_id":"2501.09775","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-do-not-understand","title":"Vision-Language Models Do Not Understand Negation","date":"2025-01-16","arxiv_id":"2501.09425","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-multilingual-llm-evaluation-for","title":"Towards Multilingual LLM Evaluation for Baltic and Nordic languages: A study on Lithuanian History","date":"2025-01-15","arxiv_id":"2501.09154","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-ai-cultural-evaluation","title":"Rethinking AI Cultural Alignment","date":"2025-01-13","arxiv_id":"2501.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-divide-and-conquer-for-fine","title":"Hierarchical Divide-and-Conquer for Fine-Grained Alignment in LLM-Based Medical Evaluation","date":"2025-01-12","arxiv_id":"2501.06741","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-token-probability-guided-rag-for","title":"First Token Probability Guided RAG for Telecom Question Answering","date":"2025-01-11","arxiv_id":"2501.06468","repositories_listed":0,"syntology":null},{"url":null,"slug":"drivingvqa-analyzing-visual-chain-of-thought","title":"DRIVINGVQA: Analyzing Visual Chain-of-Thought Reasoning of Vision Language Models in Real-World Scenarios with Driving Theory Tests","date":"2025-01-08","arxiv_id":"2501.04671","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-retrieval-based-on-generative-ai","title":"Knowledge Retrieval Based on Generative AI","date":"2025-01-08","arxiv_id":"2501.04635","repositories_listed":0,"syntology":null},{"url":null,"slug":"localizing-ai-evaluating-open-weight-language","title":"Localizing AI: Evaluating Open-Weight Language Models for Languages of Baltic States","date":"2025-01-07","arxiv_id":"2501.03952","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-up-clip-based-unanswerable-problem","title":"CLIP-UP: CLIP-Based Unanswerable Problem Detection for Visual Question Answering","date":"2025-01-02","arxiv_id":"2501.01371","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-video-llm-reasoning-via-agent-of","title":"Enhancing Video-LLM Reasoning via Agent-of-Thoughts Distillation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fsbench-a-figure-skating-benchmark-for","title":"FSBench: A Figure Skating Benchmark for Advancing Artistic Sports Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"illusionbench-a-large-scale-and-comprehensive","title":"IllusionBench: A Large-scale and Comprehensive Benchmark for Visual Illusion Understanding in Vision-Language Models","date":"2025-01-01","arxiv_id":"2501.00848","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmark-the-video-quality","title":"Q-Bench-Video: Benchmark the Video Quality Understanding of LMMs","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"separation-of-powers-on-segregating-knowledge","title":"Separation of Powers: On Segregating Knowledge from Observation in LLM-enabled Knowledge-based Visual Question Answering","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-faithfulness-metrics-for","title":"A review of faithfulness metrics for hallucination assessment in Large Language Models","date":"2024-12-31","arxiv_id":"2501.00269","repositories_listed":0,"syntology":null},{"url":null,"slug":"arastem-a-native-arabic-multiple-choice","title":"AraSTEM: A Native Arabic Multiple Choice Question Benchmark for Evaluating LLMs Knowledge In STEM Subjects","date":"2024-12-31","arxiv_id":"2501.00559","repositories_listed":0,"syntology":null},{"url":null,"slug":"equator-a-deterministic-framework-for","title":"EQUATOR: A Deterministic Framework for Evaluating LLM Reasoning with Open-Ended Questions. # v1.0.0-beta","date":"2024-12-31","arxiv_id":"2501.00257","repositories_listed":0,"syntology":null},{"url":null,"slug":"monty-hall-and-optimized-conformal-prediction","title":"Monty Hall and Optimized Conformal Prediction to Improve Decision-Making with LLMs","date":"2024-12-31","arxiv_id":"2501.00555","repositories_listed":0,"syntology":null},{"url":null,"slug":"setting-standards-in-turkish-nlp-tr-mmlu-for","title":"Setting Standards in Turkish NLP: TR-MMLU for Large Language Model Evaluation","date":"2024-12-31","arxiv_id":"2501.00593","repositories_listed":0,"syntology":null},{"url":null,"slug":"secbench-a-comprehensive-multi-dimensional","title":"SecBench: A Comprehensive Multi-Dimensional Benchmarking Dataset for LLMs in Cybersecurity","date":"2024-12-30","arxiv_id":"2412.20787","repositories_listed":0,"syntology":null},{"url":null,"slug":"hindillm-large-language-model-for-hindi","title":"HindiLLM: Large Language Model for Hindi","date":"2024-12-29","arxiv_id":"2412.20357","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-large-language-models-for-automated","title":"Using Large Language Models for Automated Grading of Student Writing about Science","date":"2024-12-25","arxiv_id":"2412.18719","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-case-you-missed-it-arc-challenge-is-not","title":"In Case You Missed It: ARC 'Challenge' Is Not That Challenging","date":"2024-12-23","arxiv_id":"2412.17758","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-you-doubtful-oh-it-might-be-difficult","title":"Are You Doubtful? Oh, It Might Be Difficult Then! Exploring the Use of Model Uncertainty for Question Difficulty Estimation","date":"2024-12-16","arxiv_id":"2412.11831","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-bidding-in-real-time-auctions-via-oracle","title":"Auto-bidding in real-time auctions via Oracle Imitation Learning (OIL)","date":"2024-12-16","arxiv_id":"2412.11434","repositories_listed":0,"syntology":null},{"url":null,"slug":"cg-bench-clue-grounded-question-answering","title":"CG-Bench: Clue-grounded Question Answering Benchmark for Long Video Understanding","date":"2024-12-16","arxiv_id":"2412.12075","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-the-forest-and-the-trees-solving","title":"Seeing the Forest and the Trees: Solving Visual Graph and Tree Based Data Structure Problems using Large Multimodal Models","date":"2024-12-15","arxiv_id":"2412.11088","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-recent-evaluation-on-the-performance-of","title":"A recent evaluation on the performance of LLMs on radiation oncology physics using questions of randomly shuffled options","date":"2024-12-14","arxiv_id":"2412.10622","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-act-as-repositories-of-causal","title":"Do LLMs Act as Repositories of Causal Knowledge?","date":"2024-12-14","arxiv_id":"2412.10635","repositories_listed":0,"syntology":null},{"url":null,"slug":"superhuman-performance-of-a-large-language","title":"Superhuman performance of a large language model on the reasoning tasks of a physician","date":"2024-12-14","arxiv_id":"2412.10849","repositories_listed":0,"syntology":null},{"url":null,"slug":"hashevict-a-pre-attention-kv-cache-eviction","title":"HashEvict: A Pre-Attention KV Cache Eviction Strategy using Locality-Sensitive Hashing","date":"2024-12-13","arxiv_id":"2412.16187","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-distillation-for-efficient-few-shot","title":"LLM Distillation for Efficient Few-Shot Multiple Choice Question Answering","date":"2024-12-13","arxiv_id":"2412.09807","repositories_listed":0,"syntology":null},{"url":null,"slug":"acq-a-unified-framework-for-automated","title":"ACQ: A Unified Framework for Automated Programmatic Creativity in Online Advertising","date":"2024-12-09","arxiv_id":"2412.06167","repositories_listed":0,"syntology":null},{"url":null,"slug":"manta-a-large-scale-multi-view-and-visual","title":"MANTA: A Large-Scale Multi-View and Visual-Text Anomaly Detection Dataset for Tiny Objects","date":"2024-12-06","arxiv_id":"2412.04867","repositories_listed":0,"syntology":null},{"url":null,"slug":"establishing-task-scaling-laws-via-compute","title":"Establishing Task Scaling Laws via Compute-Efficient Model Ladders","date":"2024-12-05","arxiv_id":"2412.04403","repositories_listed":0,"syntology":null},{"url":null,"slug":"graf-graph-retrieval-augmented-by-facts-for","title":"GRAF: Graph Retrieval Augmented by Facts for Romanian Legal Multi-Choice Question Answering","date":"2024-12-05","arxiv_id":"2412.04119","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-use-of-large-language-models-to-enhance","title":"The use of large language models to enhance cancer clinical trial educational materials","date":"2024-12-02","arxiv_id":"2412.01955","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-video-llm-via-agent-of-thoughts","title":"Unlocking Video-LLM via Agent-of-Thoughts Distillation","date":"2024-12-02","arxiv_id":"2412.01694","repositories_listed":0,"syntology":null},{"url":null,"slug":"uhura-a-benchmark-for-evaluating-scientific","title":"Uhura: A Benchmark for Evaluating Scientific Question Answering and Truthfulness in Low-Resource African Languages","date":"2024-12-01","arxiv_id":"2412.00948","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-biases-in-large-language-models-a","title":"Cognitive Biases in Large Language Models: A Survey and Mitigation Experiments","date":"2024-11-30","arxiv_id":"2412.00323","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2024-challenge-summary-and-a","title":"Perception Test 2024: Challenge Summary and a Novel Hour-Long VideoQA Benchmark","date":"2024-11-29","arxiv_id":"2411.19941","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-irt-to-distinguish-between-human-and","title":"Applying IRT to Distinguish Between Human and Generative AI Responses to Multiple-Choice Assessments","date":"2024-11-28","arxiv_id":"2412.02713","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-attention-vectors-generative","title":"Sparse Attention Vectors: Generative Multimodal Model Features Are Discriminative Vision-Language Classifiers","date":"2024-11-28","arxiv_id":"2412.00142","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-choice-learning-for-efficient-speech","title":"Multiple Choice Learning for Efficient Speech Separation with Many Speakers","date":"2024-11-27","arxiv_id":"2411.18497","repositories_listed":0,"syntology":null},{"url":null,"slug":"nemo-can-multimodal-llms-identify-attribute","title":"NEMO: Can Multimodal LLMs Identify Attribute-Modified Objects?","date":"2024-11-26","arxiv_id":"2411.17794","repositories_listed":0,"syntology":null},{"url":null,"slug":"gemex-a-large-scale-groundable-and","title":"GEMeX: A Large-Scale, Groundable, and Explainable Medical VQA Benchmark for Chest X-ray Diagnosis","date":"2024-11-25","arxiv_id":"2411.16778","repositories_listed":0,"syntology":null},{"url":null,"slug":"sageval-the-frontiers-of-satisfactory-agent","title":"SAGEval: The frontiers of Satisfactory Agent based NLG Evaluation for reference-free open-ended text","date":"2024-11-25","arxiv_id":"2411.16077","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrimed-qa-a-pan-african-multi-specialty","title":"AfriMed-QA: A Pan-African, Multi-Specialty, Medical Question-Answering Benchmark Dataset","date":"2024-11-23","arxiv_id":"2411.15640","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoautoarena-an-automated-arena-for","title":"VideoAutoArena: An Automated Arena for Evaluating Large Multimodal Models in Video Analysis through User Simulation","date":"2024-11-20","arxiv_id":"2411.13281","repositories_listed":0,"syntology":null},{"url":null,"slug":"testing-uncertainty-of-large-language-models","title":"Testing Uncertainty of Large Language Models for Physics Knowledge and Reasoning","date":"2024-11-18","arxiv_id":"2411.14465","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-benchmark-for-long-form-medical-question","title":"A Benchmark for Long-Form Medical Question Answering","date":"2024-11-14","arxiv_id":"2411.09834","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-general-to-specific-utilizing-general","title":"SHARP: Unlocking Interactive Hallucination via Stance Transfer in Role-Playing Agents","date":"2024-11-12","arxiv_id":"2411.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-consensus-through-ensemble","title":"Probabilistic Consensus through Ensemble Validation: A Framework for LLM Reliability","date":"2024-11-10","arxiv_id":"2411.06535","repositories_listed":0,"syntology":null},{"url":null,"slug":"humans-continue-to-outperform-large-language","title":"Humans and Large Language Models in Clinical Decision Support: A Study with Medical Calculators","date":"2024-11-08","arxiv_id":"2411.05897","repositories_listed":0,"syntology":null},{"url":null,"slug":"proverbeval-exploring-llm-evaluation","title":"ProverbEval: Exploring LLM Evaluation Challenges for Low-resource Language Understanding","date":"2024-11-07","arxiv_id":"2411.05049","repositories_listed":0,"syntology":null},{"url":null,"slug":"facttest-factuality-testing-in-large-language","title":"FactTest: Factuality Testing in Large Language Models with Finite-Sample and Distribution-Free Guarantees","date":"2024-11-04","arxiv_id":"2411.02603","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-evaluations-the-garbling-trick","title":"Enhancing LLM Evaluations: The Garbling Trick","date":"2024-11-03","arxiv_id":"2411.01533","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-bias-in-large-language-models","title":"Benchmarking Bias in Large Language Models during Role-Playing","date":"2024-11-01","arxiv_id":"2411.00585","repositories_listed":0,"syntology":null},{"url":null,"slug":"r-llava-improving-med-vqa-understanding","title":"R-LLaVA: Improving Med-VQA Understanding through Visual Region of Interest","date":"2024-10-27","arxiv_id":"2410.20327","repositories_listed":0,"syntology":null},{"url":"/paper/gpt-4o-system-card","slug":"gpt-4o-system-card","title":"GPT-4o System Card","date":"2024-10-25","arxiv_id":"2410.21276","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-multiple-choice-accuracy-real-world","title":"Beyond Multiple-Choice Accuracy: Real-World Challenges of Implementing Large Language Models in Healthcare","date":"2024-10-24","arxiv_id":"2410.18460","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-still-exhibit-bias-in","title":"Large Language Models Still Exhibit Bias in Long Text","date":"2024-10-23","arxiv_id":"2410.17519","repositories_listed":0,"syntology":null},{"url":null,"slug":"geocode-gpt-a-large-language-model-for","title":"GeoCode-GPT: A Large Language Model for Geospatial Code Generation Tasks","date":"2024-10-22","arxiv_id":"2410.17031","repositories_listed":0,"syntology":null},{"url":null,"slug":"susu-box-or-piggy-bank-assessing-cultural","title":"Susu Box or Piggy Bank: Assessing Cultural Commonsense Knowledge between Ghana and the U.S","date":"2024-10-21","arxiv_id":"2410.16451","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-blind-guessing-calibration-of","title":"Addressing Blind Guessing: Calibration of Selection Bias in Multiple-Choice Question Answering by Video Language Models","date":"2024-10-18","arxiv_id":"2410.14248","repositories_listed":0,"syntology":null},{"url":null,"slug":"labsafety-bench-benchmarking-llms-on-safety","title":"LabSafety Bench: Benchmarking LLMs on Safety Issues in Scientific Labs","date":"2024-10-18","arxiv_id":"2410.14182","repositories_listed":0,"syntology":null},{"url":null,"slug":"cbt-bench-evaluating-large-language-models-on","title":"CBT-Bench: Evaluating Large Language Models on Assisting Cognitive Behavior Therapy","date":"2024-10-17","arxiv_id":"2410.13218","repositories_listed":0,"syntology":null},{"url":null,"slug":"lar-echr-a-new-legal-argument-reasoning-task","title":"LAR-ECHR: A New Legal Argument Reasoning Task and Dataset for Cases of the European Court of Human Rights","date":"2024-10-17","arxiv_id":"2410.13352","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-right-and-wrong-mitigating-cold-start","title":"Not All Options Are Created Equal: Textual Option Weighting for Token-Efficient LLM-Based Knowledge Tracing","date":"2024-10-14","arxiv_id":"2410.12872","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalised-feedback-framework-for-online","title":"Personalised Feedback Framework for Online Education Programmes Using Generative AI","date":"2024-10-14","arxiv_id":"2410.11904","repositories_listed":0,"syntology":null},{"url":null,"slug":"loki-a-comprehensive-synthetic-data-detection","title":"LOKI: A Comprehensive Synthetic Data Detection Benchmark using Large Multimodal Models","date":"2024-10-13","arxiv_id":"2410.09732","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-future-of-learning-in-the-age-of","title":"The Future of Learning in the Age of Generative AI: Automated Question Generation and Assessment with Large Language Models","date":"2024-10-12","arxiv_id":"2410.09576","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrag-bench-vision-centric-evaluation-for","title":"MRAG-Bench: Vision-Centric Evaluation for Retrieval-Augmented Multimodal Models","date":"2024-10-10","arxiv_id":"2410.08182","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-then-identify-a-general-framework-for","title":"Sample then Identify: A General Framework for Risk Control and Assessment in Multimodal Large Language Models","date":"2024-10-10","arxiv_id":"2410.08174","repositories_listed":0,"syntology":null}],"record_sha256":"8bdfb5e61a19345f4a6d73818ce7cac03079f7e26fd1b633823afd6c7310c7bb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}