{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/6","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/5","next":"/task/multiple-choice/papers/7","papers":[{"url":null,"slug":"evaluating-vision-language-and-large-language","title":"Evaluating Vision-Language and Large Language Models for Automated Student Assessment in Indonesian Classrooms","date":"2025-06-05","arxiv_id":"2506.04822","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-choice-question-generation-using","title":"Multiple-Choice Question Generation Using Large Language Models: Methodology and Educator Insights","date":"2025-06-05","arxiv_id":"2506.04851","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-large-language-models-know-folktales-a","title":"Do Large Language Models Know Folktales? A Case Study of Yokai in Japanese Folktales","date":"2025-06-04","arxiv_id":"2506.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-of-leading-large-language-models","title":"Performance of leading large language models in May 2025 in Membership of the Royal College of General Practitioners-style examination questions: a cross-sectional analysis","date":"2025-06-03","arxiv_id":"2506.02987","repositories_listed":0,"syntology":null},{"url":null,"slug":"hanfu-bench-a-multimodal-benchmark-on-cross","title":"Hanfu-Bench: A Multimodal Benchmark on Cross-Temporal Cultural Understanding and Transcreation","date":"2025-06-02","arxiv_id":"2506.01565","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-multiple-choice-evaluating-steering","title":"Beyond Multiple Choice: Evaluating Steering Vectors for Adaptive Free-Form Summarization","date":"2025-05-30","arxiv_id":"2505.24859","repositories_listed":0,"syntology":null},{"url":null,"slug":"clinbench-hpb-a-clinical-benchmark-for","title":"ClinBench-HPB: A Clinical Benchmark for Evaluating LLMs in Hepato-Pancreato-Biliary Diseases","date":"2025-05-30","arxiv_id":"2506.00095","repositories_listed":0,"syntology":null},{"url":null,"slug":"persianmedqa-language-centric-evaluation-of","title":"PersianMedQA: Language-Centric Evaluation of LLMs in the Persian Medical Domain","date":"2025-05-30","arxiv_id":"2506.00250","repositories_listed":0,"syntology":null},{"url":null,"slug":"vudg-a-dataset-for-video-understanding-domain","title":"VUDG: A Dataset for Video Understanding Domain Generalization","date":"2025-05-30","arxiv_id":"2505.24346","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-aesthetic-reasoning-a-new-benchmark-for","title":"Image Aesthetic Reasoning: A New Benchmark for Medical Image Screening with MLLMs","date":"2025-05-29","arxiv_id":"2505.23265","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmsi-bench-a-benchmark-for-multi-image","title":"MMSI-Bench: A Benchmark for Multi-Image Spatial Intelligence","date":"2025-05-29","arxiv_id":"2505.23764","repositories_listed":0,"syntology":null},{"url":null,"slug":"tcm-ladder-a-benchmark-for-multimodal","title":"TCM-Ladder: A Benchmark for Multimodal Question Answering on Traditional Chinese Medicine","date":"2025-05-29","arxiv_id":"2505.24063","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-often-know-when-they","title":"Large Language Models Often Know When They Are Being Evaluated","date":"2025-05-28","arxiv_id":"2505.23836","repositories_listed":0,"syntology":null},{"url":null,"slug":"sosbench-benchmarking-safety-alignment-on","title":"SOSBENCH: Benchmarking Safety Alignment on Scientific Knowledge","date":"2025-05-27","arxiv_id":"2505.21605","repositories_listed":0,"syntology":null},{"url":null,"slug":"cp-router-an-uncertainty-aware-router-between","title":"CP-Router: An Uncertainty-Aware Router Between LLM and LRM","date":"2025-05-26","arxiv_id":"2505.19970","repositories_listed":0,"syntology":null},{"url":null,"slug":"dfir-metric-a-benchmark-dataset-for","title":"DFIR-Metric: A Benchmark Dataset for Evaluating Large Language Models in Digital Forensics and Incident Response","date":"2025-05-26","arxiv_id":"2505.19973","repositories_listed":0,"syntology":null},{"url":null,"slug":"genome-bench-a-scientific-reasoning-benchmark","title":"Genome-Bench: A Scientific Reasoning Benchmark from Real-World Expert Discussions","date":"2025-05-26","arxiv_id":"2505.19501","repositories_listed":0,"syntology":null},{"url":null,"slug":"my-answer-is-not-fair-mitigating-social-bias","title":"My Answer Is NOT 'Fair': Mitigating Social Bias in Vision-Language Models via Fair and Biased Residuals","date":"2025-05-26","arxiv_id":"2505.23798","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llms-reasoning-intensive-multimedia","title":"Enhancing LLMs' Reasoning-Intensive Multimedia Search Capabilities through Fine-Tuning and Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18831","repositories_listed":0,"syntology":null},{"url":null,"slug":"automcq-automatically-generate-code","title":"AutoMCQ -- Automatically Generate Code Comprehension Questions using GenAI","date":"2025-05-22","arxiv_id":"2505.16430","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaboration-among-multiple-large-language","title":"Collaboration among Multiple Large Language Models for Medical Question Answering","date":"2025-05-22","arxiv_id":"2505.16648","repositories_listed":0,"syntology":null},{"url":null,"slug":"kobalt-korean-benchmark-for-advanced","title":"KoBALT: Korean Benchmark For Advanced Linguistic Tasks","date":"2025-05-22","arxiv_id":"2505.16125","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-llm-first-token-predictions-in","title":"Improving LLM First-Token Predictions in Multiple-Choice Question Answering via Prefilling Attack","date":"2025-05-21","arxiv_id":"2505.15323","repositories_listed":0,"syntology":null},{"url":null,"slug":"robo2vlm-visual-question-answering-from-large","title":"Robo2VLM: Visual Question Answering from Large-Scale In-the-Wild Robot Manipulation Datasets","date":"2025-05-21","arxiv_id":"2505.15517","repositories_listed":0,"syntology":null},{"url":null,"slug":"set-llm-a-permutation-invariant-llm","title":"Set-LLM: A Permutation-Invariant LLM","date":"2025-05-21","arxiv_id":"2505.15433","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncovering-cultural-representation","title":"Uncovering Cultural Representation Disparities in Vision-Language Models","date":"2025-05-20","arxiv_id":"2505.14729","repositories_listed":0,"syntology":null},{"url":null,"slug":"wirelessmathbench-a-mathematical-modeling","title":"WirelessMathBench: A Mathematical Modeling Benchmark for LLMs in Wireless Communications","date":"2025-05-20","arxiv_id":"2505.14354","repositories_listed":0,"syntology":null},{"url":null,"slug":"lexam-benchmarking-legal-reasoning-on-340-law","title":"LEXam: Benchmarking Legal Reasoning on 340 Law Exams","date":"2025-05-19","arxiv_id":"2505.12864","repositories_listed":0,"syntology":null},{"url":null,"slug":"mr-judge-multimodal-reasoner-as-a-judge","title":"MR. Judge: Multimodal Reasoner as a Judge","date":"2025-05-19","arxiv_id":"2505.13403","repositories_listed":0,"syntology":null},{"url":null,"slug":"medguide-benchmarking-clinical-decision","title":"MedGUIDE: Benchmarking Clinical Decision-Making in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11613","repositories_listed":0,"syntology":null},{"url":null,"slug":"zerotuning-unlocking-the-initial-token-s","title":"ZeroTuning: Unlocking the Initial Token's Power to Enhance Large Language Models Without Training","date":"2025-05-16","arxiv_id":"2505.11739","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-llm-generated-plain-language-summaries","title":"Are LLM-generated plain language summaries truly understandable? A large-scale crowdsourced evaluation","date":"2025-05-15","arxiv_id":"2505.10409","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-cot-encyclopedia-analyzing-predicting-and","title":"The CoT Encyclopedia: Analyzing, Predicting, and Controlling how a Reasoning Model will Think","date":"2025-05-15","arxiv_id":"2505.10185","repositories_listed":0,"syntology":null},{"url":null,"slug":"kristeva-close-reading-as-a-novel-task-for","title":"KRISTEVA: Close Reading as a Novel Task for Benchmarking Interpretive Reasoning","date":"2025-05-14","arxiv_id":"2505.09825","repositories_listed":0,"syntology":null},{"url":null,"slug":"safepath-conformal-prediction-for-safe-llm","title":"SafePath: Conformal Prediction for Safe LLM-Based Autonomous Navigation","date":"2025-05-14","arxiv_id":"2505.09427","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-well-do-llms-reason-over-tabular-data","title":"How well do LLMs reason over tabular data, really?","date":"2025-05-12","arxiv_id":"2505.07453","repositories_listed":0,"syntology":null},{"url":null,"slug":"healthy-llms-benchmarking-llm-knowledge-of-uk","title":"Healthy LLMs? Benchmarking LLM Knowledge of UK Government Public Health Information","date":"2025-05-09","arxiv_id":"2505.06046","repositories_listed":0,"syntology":null},{"url":null,"slug":"tell-me-who-your-students-are-gpt-can","title":"Tell Me Who Your Students Are: GPT Can Generate Valid Multiple-Choice Questions When Students' (Mis)Understanding Is Hinted","date":"2025-05-09","arxiv_id":"2505.05815","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-a-framework-to-support-human","title":"Developing A Framework to Support Human Evaluation of Bias in Generated Free Response Text","date":"2025-05-05","arxiv_id":"2505.03053","repositories_listed":0,"syntology":null},{"url":"/paper/unlearning-vs-obfuscation-are-we-truly","slug":"unlearning-vs-obfuscation-are-we-truly","title":"Unlearning vs. Obfuscation: Are We Truly Removing Knowledge?","date":"2025-05-05","arxiv_id":"2505.02884","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/unlearning-vs-obfuscation-are-we-truly#ran","syntology_url":"https://syntology.ai/paper/2505.02884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02884"}},"official":null}},{"url":null,"slug":"llm-based-text-simplification-and-its-effect","title":"LLM-based Text Simplification and its Effect on User Comprehension and Cognitive Load","date":"2025-05-04","arxiv_id":"2505.01980","repositories_listed":0,"syntology":null},{"url":null,"slug":"lookalike-consistent-distractor-generation-in","title":"LookAlike: Consistent Distractor Generation in Math MCQs","date":"2025-05-03","arxiv_id":"2505.01903","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-wizard-for-removing-cross-tier","title":"Adaptive Wizard for Removing Cross-Tier Misconfigurations in Active Directory","date":"2025-05-02","arxiv_id":"2505.01028","repositories_listed":0,"syntology":null},{"url":null,"slug":"sari-structured-audio-reasoning-via","title":"SARI: Structured Audio Reasoning via Curriculum-Guided Reinforcement Learning","date":"2025-04-22","arxiv_id":"2504.15900","repositories_listed":0,"syntology":null},{"url":null,"slug":"longperceptualthoughts-distilling-system-2","title":"LongPerceptualThoughts: Distilling System-2 Reasoning for System-1 Perception","date":"2025-04-21","arxiv_id":"2504.15362","repositories_listed":0,"syntology":null},{"url":null,"slug":"farseval-pkbets-a-new-diverse-benchmark-for","title":"FarsEval-PKBETS: A new diverse benchmark for evaluating Persian large language models","date":"2025-04-20","arxiv_id":"2504.14690","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-ai-generated-questions-alignment","title":"Assessing AI-Generated Questions' Alignment with Cognitive Frameworks in Educational Assessment","date":"2025-04-19","arxiv_id":"2504.14232","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-gen-automatic-distractor-generation-and","title":"D-GEN: Automatic Distractor Generation and Evaluation for Reliable Assessment of Generative Model","date":"2025-04-18","arxiv_id":"2504.13439","repositories_listed":0,"syntology":null},{"url":null,"slug":"dmind-benchmark-toward-a-holistic-assessment","title":"DMind Benchmark: Toward a Holistic Assessment of LLM Capabilities across the Web3 Domain","date":"2025-04-18","arxiv_id":"2504.16116","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-next-generation-reasoning","title":"Benchmarking Next-Generation Reasoning-Focused Large Language Models in Ophthalmology: A Head-to-Head Evaluation on 5,888 Items","date":"2025-04-15","arxiv_id":"2504.11186","repositories_listed":0,"syntology":null},{"url":null,"slug":"agmmu-a-comprehensive-agricultural-multimodal","title":"AgMMU: A Comprehensive Agricultural Multimodal Understanding and Reasoning Benchmark","date":"2025-04-14","arxiv_id":"2504.10568","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-could-be-rote-learners","title":"Large Language Models Could Be Rote Learners","date":"2025-04-11","arxiv_id":"2504.08300","repositories_listed":0,"syntology":null},{"url":null,"slug":"instructionbench-an-instructional-video","title":"InstructionBench: An Instructional Video Understanding Benchmark","date":"2025-04-07","arxiv_id":"2504.05040","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-ai-master-construction-management-cm","title":"Can AI Master Construction Management (CM)? Benchmarking State-of-the-Art Large Language Models on CM Certification Exams","date":"2025-04-04","arxiv_id":"2504.08779","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-chatgpt-to-deepseek-ai-a-comprehensive","title":"From ChatGPT to DeepSeek AI: A Comprehensive Analysis of Evolution, Deviation, and Future Implications in AI-Language Models","date":"2025-04-04","arxiv_id":"2504.03219","repositories_listed":0,"syntology":null},{"url":null,"slug":"acpbench-hard-unrestrained-reasoning-about","title":"ACPBench Hard: Unrestrained Reasoning about Action, Change, and Planning","date":"2025-03-31","arxiv_id":"2503.24378","repositories_listed":0,"syntology":null},{"url":null,"slug":"order-independence-with-finetuning","title":"Order Independence With Finetuning","date":"2025-03-30","arxiv_id":"2503.23483","repositories_listed":0,"syntology":null},{"url":null,"slug":"unmasking-deceptive-visuals-benchmarking","title":"Unmasking Deceptive Visuals: Benchmarking Multimodal Large Language Models on Misleading Chart Question Answering","date":"2025-03-23","arxiv_id":"2503.18172","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpbench-a-comprehensive-and-fine-grained","title":"Evaluating Clinical Competencies of Large Language Models with a General Practice Benchmark","date":"2025-03-22","arxiv_id":"2503.17599","repositories_listed":0,"syntology":null},{"url":null,"slug":"saudiculture-a-benchmark-for-evaluating-large","title":"SaudiCulture: A Benchmark for Evaluating Large Language Models Cultural Competence within Saudi Arabia","date":"2025-03-21","arxiv_id":"2503.17485","repositories_listed":0,"syntology":null},{"url":null,"slug":"autodrive-qa-automated-generation-of-multiple","title":"AutoDrive-QA- Automated Generation of Multiple-Choice Questions for Autonomous Driving Datasets Using Large Vision-Language Models","date":"2025-03-20","arxiv_id":"2503.15778","repositories_listed":0,"syntology":null},{"url":null,"slug":"codereviewqa-the-code-review-comprehension","title":"CodeReviewQA: The Code Review Comprehension Assessment for Large Language Models","date":"2025-03-20","arxiv_id":"2503.16167","repositories_listed":0,"syntology":null},{"url":null,"slug":"favor-bench-a-comprehensive-benchmark-for","title":"FAVOR-Bench: A Comprehensive Benchmark for Fine-Grained Video Motion Understanding","date":"2025-03-19","arxiv_id":"2503.14935","repositories_listed":0,"syntology":null},{"url":null,"slug":"visnumbench-evaluating-number-sense-of","title":"VisNumBench: Evaluating Number Sense of Multimodal Large Language Models","date":"2025-03-19","arxiv_id":"2503.14939","repositories_listed":0,"syntology":null},{"url":null,"slug":"chat-ts-enhancing-multi-modal-reasoning-over","title":"Chat-TS: Enhancing Multi-Modal Reasoning Over Time-Series and Natural Language Data","date":"2025-03-13","arxiv_id":"2503.10883","repositories_listed":0,"syntology":null},{"url":null,"slug":"it-is-too-many-options-pitfalls-of-multiple","title":"It is Too Many Options: Pitfalls of Multiple-Choice Questions in Generative AI and Medical Education","date":"2025-03-13","arxiv_id":"2503.13508","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-item-writing-flaws-on","title":"The Impact of Item-Writing Flaws on Difficulty and Discrimination in Item Response Theory","date":"2025-03-13","arxiv_id":"2503.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"identity-lock-locking-api-fine-tuned-llms","title":"Identity Lock: Locking API Fine-tuned LLMs With Identity-based Wake Words","date":"2025-03-10","arxiv_id":"2503.10668","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-bias-benchmark-for-generation-a","title":"Social Bias Benchmark for Generation: A Comparison of Generation and QA-Based Evaluations","date":"2025-03-10","arxiv_id":"2503.06987","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-conversational-ai-for-disease","title":"Towards Conversational AI for Disease Management","date":"2025-03-08","arxiv_id":"2503.06074","repositories_listed":0,"syntology":null},{"url":null,"slug":"urbanvideo-bench-benchmarking-vision-language","title":"UrbanVideo-Bench: Benchmarking Vision-Language Models on Embodied Intelligence with Video Data in Urban Spaces","date":"2025-03-08","arxiv_id":"2503.06157","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-guarantees-of-correctness","title":"Correctness Coverage Evaluation for Medical Multiple-Choice Question Answering Based on the Enhanced Conformal Prediction Framework","date":"2025-03-07","arxiv_id":"2503.05505","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-outputs-enable-general-purpose","title":"Structured Outputs Enable General-Purpose LLMs to be Medical Experts","date":"2025-03-05","arxiv_id":"2503.03194","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-ai-and-peer-feedback-on","title":"The impact of AI and peer feedback on research writing skills: a study using the CGScholar platform among Kazakhstani scholars","date":"2025-03-05","arxiv_id":"2503.05820","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-above-less-of-the-right-parallel","title":"None of the Above, Less of the Right: Parallel Patterns between Humans and LLMs on Multi-Choice Questions Answering","date":"2025-03-03","arxiv_id":"2503.01550","repositories_listed":0,"syntology":null},{"url":null,"slug":"mv-math-evaluating-multimodal-math-reasoning","title":"MV-MATH: Evaluating Multimodal Math Reasoning in Multi-Visual Contexts","date":"2025-02-28","arxiv_id":"2502.20808","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-rlvr-emerging-medical-reasoning-from-a-3b","title":"Med-RLVR: Emerging Medical Reasoning from a 3B base model via reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19655","repositories_listed":0,"syntology":null},{"url":null,"slug":"anpmi-assessing-the-true-comprehension","title":"ANPMI: Assessing the True Comprehension Capabilities of LLMs for Multiple Choice Questions","date":"2025-02-26","arxiv_id":"2502.18798","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepseek-r1-outperforms-gemini-2-0-pro-openai","title":"DeepSeek-R1 Outperforms Gemini 2.0 Pro, OpenAI o1, and o3-mini in Bilingual Complex Ophthalmology Reasoning","date":"2025-02-25","arxiv_id":"2502.17947","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversal-blessing-thinking-backward-may","title":"Reversal Blessing: Thinking Backward May Outpace Thinking Forward in Multi-choice Questions","date":"2025-02-25","arxiv_id":"2502.18435","repositories_listed":0,"syntology":null},{"url":null,"slug":"secura-sigmoid-enhanced-cur-decomposition","title":"SECURA: Sigmoid-Enhanced CUR Decomposition with Uninterrupted Retention and Low-Rank Adaptation in Large Language Models","date":"2025-02-25","arxiv_id":"2502.18168","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lazy-student-s-dream-chatgpt-passing-an","title":"The Lazy Student's Dream: ChatGPT Passing an Engineering Course on Its Own","date":"2025-02-23","arxiv_id":"2503.05760","repositories_listed":0,"syntology":null},{"url":null,"slug":"legalbench-pt-a-benchmark-for-portuguese-law","title":"LegalBench.PT: A Benchmark for Portuguese Law","date":"2025-02-22","arxiv_id":"2502.16357","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-make-mistakes-like-students-exploring","title":"Do LLMs Make Mistakes Like Students? Exploring Natural Alignment between Language Models and Human Error Patterns","date":"2025-02-21","arxiv_id":"2502.15140","repositories_listed":0,"syntology":null},{"url":null,"slug":"mhqa-a-diverse-knowledge-intensive-mental","title":"MHQA: A Diverse, Knowledge Intensive Mental Health Question Answering Challenge for Language Models","date":"2025-02-21","arxiv_id":"2502.15418","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundamental-limitations-in-defending-llm","title":"Fundamental Limitations in Defending LLM Finetuning APIs","date":"2025-02-20","arxiv_id":"2502.14828","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcqa-eval-efficient-confidence-evaluation-in","title":"MCQA-Eval: Efficient Confidence Evaluation in NLG with Gold-Standard Correctness Labels","date":"2025-02-20","arxiv_id":"2502.14268","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-cultural-blind-spots-analyzing-the","title":"Unveiling Cultural Blind Spots: Analyzing the Limitations of mLLMs in Procedural Text Comprehension","date":"2025-02-20","arxiv_id":"2502.14315","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-tuning-on-public-government-and","title":"Instruction Tuning on Public Government and Cultural Data for Low-Resource Language: a Case Study in Kazakh","date":"2025-02-19","arxiv_id":"2502.13647","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-this-collection-worth-my-llm-s-time","title":"Is This Collection Worth My LLM's Time? Automatically Measuring Information Potential in Text Corpora","date":"2025-02-19","arxiv_id":"2502.13691","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-geo-culturally-grounded-llm","title":"Towards Geo-Culturally Grounded LLM Generations","date":"2025-02-19","arxiv_id":"2502.13497","repositories_listed":0,"syntology":null},{"url":null,"slug":"vital-a-new-dataset-for-benchmarking","title":"VITAL: A New Dataset for Benchmarking Pluralistic Alignment in Healthcare","date":"2025-02-19","arxiv_id":"2502.13775","repositories_listed":0,"syntology":null},{"url":null,"slug":"which-of-these-best-describes-multiple-choice","title":"Which of These Best Describes Multiple Choice Evaluation with LLMs? A) Forced B) Flawed C) Fixable D) All of the Above","date":"2025-02-19","arxiv_id":"2502.14127","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-profile-from-surface-level-facts-to","title":"Beyond Profile: From Surface-Level Facts to Deep Persona Simulation in LLMs","date":"2025-02-18","arxiv_id":"2502.12988","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-others-a-general-technique-to","title":"None of the Others: a General Technique to Distinguish Reasoning from Memorization in Multiple-Choice LLM Evaluation Benchmarks","date":"2025-02-18","arxiv_id":"2502.12896","repositories_listed":0,"syntology":null},{"url":null,"slug":"occult-evaluating-large-language-models-for","title":"OCCULT: Evaluating Large Language Models for Offensive Cyber Operation Capabilities","date":"2025-02-18","arxiv_id":"2502.15797","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-retrieval-augmentation-for-open","title":"Multi-Modal Retrieval Augmentation for Open-Ended and Knowledge-Intensive Video Question Answering","date":"2025-02-17","arxiv_id":"2502.11747","repositories_listed":0,"syntology":null},{"url":null,"slug":"logidynamics-unraveling-the-dynamics-of","title":"LogiDynamics: Unraveling the Dynamics of Logical Inference in Large Language Model Reasoning","date":"2025-02-16","arxiv_id":"2502.11176","repositories_listed":0,"syntology":null},{"url":"/paper/viscon-100k-leveraging-contextual-web-data","slug":"viscon-100k-leveraging-contextual-web-data","title":"VisCon-100K: Leveraging Contextual Web Data for Fine-tuning Vision Language Models","date":"2025-02-14","arxiv_id":"2502.10250","repositories_listed":0,"syntology":null},{"url":null,"slug":"objective-quantification-of-mood-states-using","title":"Objective quantification of mood states using large language models","date":"2025-02-13","arxiv_id":"2502.09487","repositories_listed":0,"syntology":null}],"record_sha256":"b93f34c21e95689d893f13a0d51aeade2ffe7c999d877e56e27a26275973f66e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}