{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/8","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":12,"rows_per_page":100,"rows":[701,800],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/7","next":"/task/multiple-choice/papers/9","papers":[{"url":"/paper/tvbench-redesigning-video-language-evaluation","slug":"tvbench-redesigning-video-language-evaluation","title":"TVBench: Redesigning Video-Language Evaluation","date":"2024-10-10","arxiv_id":"2410.07752","repositories_listed":0,"syntology":null},{"url":null,"slug":"answering-questions-in-stages-prompt-chaining","title":"Answering Questions in Stages: Prompt Chaining for Contract QA","date":"2024-10-09","arxiv_id":"2410.12840","repositories_listed":0,"syntology":null},{"url":null,"slug":"acpbench-reasoning-about-action-change-and","title":"ACPBench: Reasoning about Action, Change, and Planning","date":"2024-10-08","arxiv_id":"2410.05669","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionatlas-a-videoqa-benchmark-for-domain","title":"ActionAtlas: A VideoQA Benchmark for Domain-specialized Action Recognition","date":"2024-10-08","arxiv_id":"2410.05774","repositories_listed":0,"syntology":null},{"url":null,"slug":"listening-to-the-wise-few-select-and-copy","title":"Listening to the Wise Few: Select-and-Copy Attention Heads for Multiple-Choice QA","date":"2024-10-03","arxiv_id":"2410.02343","repositories_listed":0,"syntology":null},{"url":"/paper/video-instruction-tuning-with-synthetic-data","slug":"video-instruction-tuning-with-synthetic-data","title":"Video Instruction Tuning With Synthetic Data","date":"2024-10-03","arxiv_id":"2410.02713","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-enhanced-model-for-eye-leme-an-open","title":"Language Enhanced Model for Eye (LEME): An Open-Source Ophthalmology-Specific Large Language Model","date":"2024-10-01","arxiv_id":"2410.03740","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-bench-video-benchmarking-the-video-quality","title":"Q-Bench-Video: Benchmarking the Video Quality Understanding of LMMs","date":"2024-09-30","arxiv_id":"2409.20063","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-adaptive-pretrained-language-models-via","title":"Task-Adaptive Pretrained Language Models via Clustered-Importance Sampling","date":"2024-09-30","arxiv_id":"2410.03735","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-selection-bias-with-node-pruning","title":"Mitigating Selection Bias with Node Pruning and Auxiliary Options","date":"2024-09-27","arxiv_id":"2409.18857","repositories_listed":0,"syntology":null},{"url":null,"slug":"dare-diverse-visual-question-answering-with","title":"DARE: Diverse Visual Question Answering with Robustness Evaluation","date":"2024-09-26","arxiv_id":"2409.18023","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-sciq-an-educational-chatbot-for","title":"LLaMa-SciQ: An Educational Chatbot for Answering Science MCQ","date":"2024-09-25","arxiv_id":"2409.16779","repositories_listed":0,"syntology":null},{"url":null,"slug":"riscore-enhancing-in-context-riddle-solving","title":"RISCORE: Enhancing In-Context Riddle Solving in Language Models through Context-Reconstructed Example Augmentation","date":"2024-09-24","arxiv_id":"2409.16383","repositories_listed":0,"syntology":null},{"url":null,"slug":"detect-describe-discriminate-moving-beyond","title":"Detect, Describe, Discriminate: Moving Beyond VQA for MLLM Evaluation","date":"2024-09-23","arxiv_id":"2409.15125","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-performance-and-robustness-of","title":"Evaluating the Performance and Robustness of LLMs in Materials Science Q&A and Property Predictions","date":"2024-09-22","arxiv_id":"2409.14572","repositories_listed":0,"syntology":null},{"url":null,"slug":"first-place-solution-to-the-multiple-choice","title":"First Place Solution to the Multiple-choice Video QA Track of The Second Perception Test Challenge","date":"2024-09-20","arxiv_id":"2409.13538","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilingual-evaluation-of-language-models-on","title":"Bilingual Evaluation of Language Models on General Knowledge in University Entrance Exams with Minimal Contamination","date":"2024-09-19","arxiv_id":"2409.12746","repositories_listed":0,"syntology":null},{"url":null,"slug":"edu-values-towards-evaluating-the-chinese","title":"Edu-Values: Towards Evaluating the Chinese Education Values of Large Language Models","date":"2024-09-19","arxiv_id":"2409.12739","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-knowledge-distillation-empowering","title":"Efficient Knowledge Distillation: Empowering Small Language Models with Teacher Model Insights","date":"2024-09-19","arxiv_id":"2409.12586","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-as-a-judge-reward-model-what-they-can-and","title":"LLM-as-a-Judge & Reward Model: What They Can and Cannot Do","date":"2024-09-17","arxiv_id":"2409.11239","repositories_listed":0,"syntology":null},{"url":null,"slug":"cracking-the-code-multi-domain-llm-evaluation","title":"Cracking the Code: Multi-domain LLM Evaluation on Real-World Professional Exams in Indonesia","date":"2024-09-13","arxiv_id":"2409.08564","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-syntactic-information-in-sentence","title":"Exploring syntactic information in sentence embeddings through multilingual subject-verb agreement","date":"2024-09-10","arxiv_id":"2409.06567","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-democratizing-multilingual-large","title":"Towards Democratizing Multilingual Large Language Models For Medicine Through A Two-Stage Instruction Fine-tuning Approach","date":"2024-09-09","arxiv_id":"2409.05732","repositories_listed":0,"syntology":null},{"url":null,"slug":"materialbench-evaluating-college-level","title":"MaterialBENCH: Evaluating College-Level Materials Science Problem-Solving Abilities of Large Language Models","date":"2024-09-05","arxiv_id":"2409.03161","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-role-of-large-language-models-in","title":"The Role of Large Language Models in Musicology: Are We Ready to Trust the Machines?","date":"2024-09-03","arxiv_id":"2409.01864","repositories_listed":0,"syntology":null},{"url":null,"slug":"novel-wd-exploring-acquisition-of-novel-world","title":"Novel-WD: Exploring acquisition of Novel World Knowledge in LLMs Using Prefix-Tuning","date":"2024-08-30","arxiv_id":"2408.17070","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-are-self-taught","title":"Large Language Models Are Self-Taught Reasoners: Enhancing LLM Applications via Tailored Problem-Solving Demonstrations","date":"2024-08-22","arxiv_id":"2408.12315","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-susceptible-are-llms-to-influence-in","title":"How Susceptible are LLMs to Influence in Prompts?","date":"2024-08-17","arxiv_id":"2408.11865","repositories_listed":0,"syntology":null},{"url":null,"slug":"examining-the-behavior-of-llm-architectures","title":"Examining the Behavior of LLM Architectures Within the Framework of Standardized National Exams in Brazil","date":"2024-08-09","arxiv_id":"2408.05035","repositories_listed":0,"syntology":null},{"url":null,"slug":"winning-amazon-kdd-cup-24","title":"Winning Amazon KDD Cup'24","date":"2024-08-05","arxiv_id":"2408.04658","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02114","title":"Recent Advances in Multi-Choice Machine Reading Comprehension: A Survey on Methods and Datasets","date":"2024-08-04","arxiv_id":"2408.02114","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-know-when-to-not-answer-investigating","title":"Do LLMs Know When to NOT Answer? Investigating Abstention Abilities of Large Language Models","date":"2024-07-23","arxiv_id":"2407.16221","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-few-shot-image-classification","title":"Improved Few-Shot Image Classification Through Multiple-Choice Questions","date":"2024-07-23","arxiv_id":"2407.16145","repositories_listed":0,"syntology":null},{"url":"/paper/answer-assemble-ace-understanding-how","slug":"answer-assemble-ace-understanding-how","title":"Answer, Assemble, Ace: Understanding How Transformers Answer Multiple Choice Questions","date":"2024-07-21","arxiv_id":"2407.15018","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-assemble-ace-understanding-how#ran","syntology_url":"https://syntology.ai/paper/2407.15018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15018"}},"official":null}},{"url":null,"slug":"mibench-evaluating-multimodal-large-language","title":"MIBench: Evaluating Multimodal Large Language Models over Multiple Images","date":"2024-07-21","arxiv_id":"2407.15272","repositories_listed":0,"syntology":null},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":null,"slug":"adversarial-databases-improve-success-in","title":"Adversarial Databases Improve Success in Retrieval-based Large Language Models","date":"2024-07-19","arxiv_id":"2407.14609","repositories_listed":0,"syntology":null},{"url":null,"slug":"mini-llm-memory-efficient-structured-pruning","title":"MINI-LLM: Memory-Efficient Structured Pruning for Large Language Models","date":"2024-07-16","arxiv_id":"2407.11681","repositories_listed":0,"syntology":null},{"url":null,"slug":"astromlab-1-who-wins-astronomy-jeopardy","title":"AstroMLab 1: Who Wins Astronomy Jeopardy!?","date":"2024-07-15","arxiv_id":"2407.11194","repositories_listed":0,"syntology":null},{"url":null,"slug":"ntsebench-cognitive-reasoning-benchmark-for","title":"NTSEBENCH: Cognitive Reasoning Benchmark for Vision Language Models","date":"2024-07-15","arxiv_id":"2407.10380","repositories_listed":0,"syntology":null},{"url":null,"slug":"lab-bench-measuring-capabilities-of-language","title":"LAB-Bench: Measuring Capabilities of Language Models for Biology Research","date":"2024-07-14","arxiv_id":"2407.10362","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-nuanced-bias-in-large-language","title":"Evaluating Nuanced Bias in Large Language Model Free Response Answers","date":"2024-07-11","arxiv_id":"2407.08842","repositories_listed":0,"syntology":null},{"url":null,"slug":"cfinbench-a-comprehensive-chinese-financial","title":"CFinBench: A Comprehensive Chinese Financial Benchmark for Large Language Models","date":"2024-07-02","arxiv_id":"2407.02301","repositories_listed":0,"syntology":null},{"url":null,"slug":"changing-answer-order-can-decrease-mmlu","title":"Changing Answer Order Can Decrease MMLU Accuracy","date":"2024-06-27","arxiv_id":"2406.19470","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-visual-and-cultural-interpretation","title":"Evaluating Visual and Cultural Interpretation: The K-Viscuit Benchmark with Human-VLM Collaboration","date":"2024-06-24","arxiv_id":"2406.16469","repositories_listed":0,"syntology":null},{"url":null,"slug":"syndarin-synthesising-datasets-for-automated","title":"SynDARin: Synthesising Datasets for Automated Reasoning in Low-Resource Languages","date":"2024-06-20","arxiv_id":"2406.14425","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-distractor-generation-for-multiple","title":"Enhancing Distractor Generation for Multiple-Choice Questions with Retrieval Augmented Pretraining and Knowledge Graph Integration","date":"2024-06-19","arxiv_id":"2406.13578","repositories_listed":0,"syntology":null},{"url":null,"slug":"qrmem-unleash-the-length-limitation-through","title":"QRMeM: Unleash the Length Limitation through Question then Reflection Memory Mechanism","date":"2024-06-19","arxiv_id":"2406.13167","repositories_listed":0,"syntology":null},{"url":null,"slug":"aqulia-med-llm-pioneering-full-process-open","title":"Aqulia-Med LLM: Pioneering Full-Process Open-Source Medical Language Models","date":"2024-06-18","arxiv_id":"2406.12182","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-principles-behind-opinion-dynamics-in","title":"On the Principles behind Opinion Dynamics in Multi-Agent Systems of Large Language Models","date":"2024-06-18","arxiv_id":"2406.15492","repositories_listed":0,"syntology":null},{"url":null,"slug":"qog-question-and-options-generation-based-on","title":"QOG:Question and Options Generation based on Language Model","date":"2024-06-18","arxiv_id":"2406.12381","repositories_listed":0,"syntology":null},{"url":null,"slug":"velociti-can-video-language-models-bind","title":"VELOCITI: Benchmarking Video-Language Compositional Reasoning with Strict Entailment","date":"2024-06-16","arxiv_id":"2406.10889","repositories_listed":0,"syntology":null},{"url":null,"slug":"vceval-rethinking-what-is-a-good-educational","title":"VCEval: Rethinking What is a Good Educational Video and How to Automatically Evaluate It","date":"2024-06-15","arxiv_id":"2407.12005","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignmmbench-evaluating-chinese-multimodal","title":"AlignMMBench: Evaluating Chinese Multimodal Alignment in Large Vision-Language Models","date":"2024-06-13","arxiv_id":"2406.09295","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-statistical-modeling-with-predictors","title":"Bayesian Statistical Modeling with Predictors from LLMs","date":"2024-06-13","arxiv_id":"2406.09012","repositories_listed":0,"syntology":null},{"url":null,"slug":"olmes-a-standard-for-language-model","title":"OLMES: A Standard for Language Model Evaluations","date":"2024-06-12","arxiv_id":"2406.08446","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-making-behavior-evaluation-framework","title":"Decision-Making Behavior Evaluation Framework for LLMs under Uncertain Context","date":"2024-06-10","arxiv_id":"2406.05972","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-personal-health-large-language","title":"Towards a Personal Health Large Language Model","date":"2024-06-10","arxiv_id":"2406.06474","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-recognize-me-when-i-is-not-me","title":"Do LLMs Recognize me, When I is not me: Assessment of LLMs Understanding of Turkish Indexical Pronouns in Indexical Shift Contexts","date":"2024-06-08","arxiv_id":"2406.05569","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-and-addressing-hallucinations","title":"Investigating and Addressing Hallucinations of LLMs in Tasks Involving Negation","date":"2024-06-08","arxiv_id":"2406.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"criskeval-a-chinese-multi-level-risk","title":"CRiskEval: A Chinese Multi-Level Risk Evaluation Benchmark Dataset for Large Language Models","date":"2024-06-07","arxiv_id":"2406.04752","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-has-predicting-downstream-capabilities-of","title":"Why Has Predicting Downstream Capabilities of Frontier AI Models with Scale Remained Elusive?","date":"2024-06-06","arxiv_id":"2406.04391","repositories_listed":0,"syntology":null},{"url":null,"slug":"explore-then-determine-a-gnn-llm-synergy","title":"Explore then Determine: A GNN-LLM Synergy Framework for Reasoning over Knowledge Graph","date":"2024-06-03","arxiv_id":"2406.01145","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgrc-an-effective-fine-tuning-framework-for","title":"DGRC: An Effective Fine-tuning Framework for Distractor Generation in Chinese Multi-choice Reading Comprehension","date":"2024-05-29","arxiv_id":"2405.19139","repositories_listed":0,"syntology":null},{"url":null,"slug":"edinburgh-clinical-nlp-at-mediqa-corr-2024","title":"Edinburgh Clinical NLP at MEDIQA-CORR 2024: Guiding Large Language Models with Hints","date":"2024-05-28","arxiv_id":"2405.18028","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-trust-llms-mitigate-overconfidence","title":"Can We Trust LLMs? Mitigate Overconfidence Bias in LLMs through Knowledge Transfer","date":"2024-05-27","arxiv_id":"2405.16856","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagery-as-inquiry-exploring-a-multimodal","title":"Imagery as Inquiry: Exploring A Multimodal Dataset for Conversational Recommendation","date":"2024-05-23","arxiv_id":"2405.14142","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-portfolio-optimization-model-for","title":"Robust portfolio optimization model for electronic coupon allocation","date":"2024-05-21","arxiv_id":"2405.12865","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-capabilities-of-prompted-large","title":"Exploring the Capabilities of Prompted Large Language Models in Educational and Assessment Applications","date":"2024-05-19","arxiv_id":"2405.11579","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognet-md-an-evaluation-framework-and-dataset","title":"COGNET-MD, an evaluation framework and dataset for Large Language Model benchmarks in the medical domain","date":"2024-05-17","arxiv_id":"2405.10893","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-generalist-to-specialist-improving-large","title":"From Generalist to Specialist: Improving Large Language Models for Medical Physics Using ARCoT","date":"2024-05-17","arxiv_id":"2405.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"amazutah-nlp-at-semeval-2024-task-9-a","title":"AmazUtah_NLP at SemEval-2024 Task 9: A MultiChoice Question Answering System for Commonsense Defying Reasoning","date":"2024-05-16","arxiv_id":"2405.10385","repositories_listed":0,"syntology":null},{"url":"/paper/cinepile-a-long-video-question-answering","slug":"cinepile-a-long-video-question-answering","title":"CinePile: A Long Video Question Answering Dataset and Benchmark","date":"2024-05-14","arxiv_id":"2405.08813","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcs-sql-leveraging-multiple-prompts-and","title":"MCS-SQL: Leveraging Multiple Prompts and Multiple-Choice Selection For Text-to-SQL Generation","date":"2024-05-13","arxiv_id":"2405.07467","repositories_listed":0,"syntology":null},{"url":null,"slug":"worldqa-multimodal-world-knowledge-in-videos","title":"WorldQA: Multimodal World Knowledge in Videos through Long-Chain Reasoning","date":"2024-05-06","arxiv_id":"2405.03272","repositories_listed":0,"syntology":null},{"url":null,"slug":"math-multiple-choice-question-generation-via","title":"Math Multiple Choice Question Generation via Human-Large Language Model Collaboration","date":"2024-05-01","arxiv_id":"2405.00864","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundabench-evaluating-chinese-fundamental","title":"FoundaBench: Evaluating Chinese Fundamental Knowledge Capabilities of Large Language Models","date":"2024-04-29","arxiv_id":"2404.18359","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-and-machine-learning-for-next-generation","title":"AI and Machine Learning for Next Generation Science Assessments","date":"2024-04-23","arxiv_id":"2405.06660","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-automated-distractor-generation-for","title":"Improving Automated Distractor Generation for Math Multiple-choice Questions with Overgenerate-and-rank","date":"2024-04-19","arxiv_id":"2405.05144","repositories_listed":0,"syntology":null},{"url":"/paper/blink-multimodal-large-language-models-can","slug":"blink-multimodal-large-language-models-can","title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","date":"2024-04-18","arxiv_id":"2404.12390","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucibot-is-there-no-such-thing-as-a-bad","title":"Is There No Such Thing as a Bad Question? H4R: HalluciBot For Ratiocination, Rewriting, Ranking, and Routing","date":"2024-04-18","arxiv_id":"2404.12535","repositories_listed":0,"syntology":null},{"url":null,"slug":"villm-eval-a-comprehensive-evaluation-suite","title":"ViLLM-Eval: A Comprehensive Evaluation Suite for Vietnamese Large Language Models","date":"2024-04-17","arxiv_id":"2404.11086","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-difficulty-ranking-for-multiple","title":"Question Difficulty Ranking for Multiple-Choice Reading Comprehension","date":"2024-04-16","arxiv_id":"2404.10704","repositories_listed":0,"syntology":null},{"url":"/paper/morevqa-exploring-modular-reasoning-models","slug":"morevqa-exploring-modular-reasoning-models","title":"MoReVQA: Exploring Modular Reasoning Models for Video Question Answering","date":"2024-04-09","arxiv_id":"2404.06511","repositories_listed":0,"syntology":null},{"url":null,"slug":"cleared-for-takeoff-compositional-conditional","title":"Cleared for Takeoff? Compositional & Conditional Reasoning may be the Achilles Heel to (Flight-Booking) Language Agents","date":"2024-04-05","arxiv_id":"2404.04237","repositories_listed":0,"syntology":null},{"url":null,"slug":"lhmke-a-large-scale-holistic-multi-subject","title":"LHMKE: A Large-scale Holistic Multi-subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2024-03-19","arxiv_id":"2403.12601","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-event-causality-identification-with","title":"Enhancing Event Causality Identification with Rationale and Structure-Aware Causal Question Answering","date":"2024-03-17","arxiv_id":"2403.11129","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-image-classification-and","title":"Few-Shot Image Classification and Segmentation as Visual Question Answering Using Vision-Language Models","date":"2024-03-15","arxiv_id":"2403.10287","repositories_listed":0,"syntology":null},{"url":null,"slug":"aratrust-an-evaluation-of-trustworthiness-for","title":"AraTrust: An Evaluation of Trustworthiness for LLMs in Arabic","date":"2024-03-14","arxiv_id":"2403.09017","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-comprehension-of-chatgpt-in","title":"Exploring the Comprehension of ChatGPT in Traditional Chinese Medicine Knowledge","date":"2024-03-14","arxiv_id":"2403.09164","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-generative-large-language-model","title":"Rethinking Generative Large Language Model Evaluation for Semantic Comprehension","date":"2024-03-12","arxiv_id":"2403.07872","repositories_listed":0,"syntology":null},{"url":null,"slug":"medkp-medical-dialogue-with-knowledge","title":"MedKP: Medical Dialogue with Knowledge Enhancement and Clinical Pathway Encoding","date":"2024-03-11","arxiv_id":"2403.06611","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-traditional-chinese-evaluation","title":"An Improved Traditional Chinese Evaluation Suite for Foundation Model","date":"2024-03-04","arxiv_id":"2403.01858","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-generation-of-multiple-choice-cloze","title":"Automated Generation of Multiple-Choice Cloze Questions for Assessing English Vocabulary Using GPT-turbo 3.5","date":"2024-03-04","arxiv_id":"2403.02078","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-cloze-test-question-item","title":"Controlling Cloze-test Question Item Difficulty with PLM-based Surrogate Models for IRT Assessment","date":"2024-03-03","arxiv_id":"2403.01456","repositories_listed":0,"syntology":null},{"url":null,"slug":"kormedmcqa-multi-choice-question-answering","title":"KorMedMCQA: Multi-Choice Question Answering Benchmark for Korean Healthcare Professional Licensing Examinations","date":"2024-03-03","arxiv_id":"2403.01469","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictions-from-language-models-for-multiple","title":"Predictions from language models for multiple-choice tasks are not robust under variation of scoring methods","date":"2024-03-01","arxiv_id":"2403.00998","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-multiple-choices-question","title":"Unsupervised multiple choices question answering via universal corpus","date":"2024-02-27","arxiv_id":"2402.17333","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-multiple-personalities-in-large","title":"Identifying Multiple Personalities in Large Language Models with External Evaluation","date":"2024-02-22","arxiv_id":"2402.14805","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-probabilities-unveiling-the","title":"Beyond Probabilities: Unveiling the Misalignment in Evaluating Large Language Models","date":"2024-02-21","arxiv_id":"2402.13887","repositories_listed":0,"syntology":null}],"record_sha256":"161f55b32e0a092d4e2f82bd4c7e859ea062a58dc91bfd54d241bc1263f658c3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}