{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/2","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":12,"rows_per_page":100,"rows":[101,200],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice","next":"/task/multiple-choice/papers/3","papers":[{"url":"/paper/mobile-mmlu-a-mobile-intelligence-language","slug":"mobile-mmlu-a-mobile-intelligence-language","title":"Mobile-MMLU: A Mobile Intelligence Language Understanding Benchmark","date":"2025-03-26","arxiv_id":"2503.20786","repositories_listed":1,"syntology":null},{"url":"/paper/language-model-uncertainty-quantification","slug":"language-model-uncertainty-quantification","title":"Language Model Uncertainty Quantification with Attention Chain","date":"2025-03-24","arxiv_id":"2503.19168","repositories_listed":1,"syntology":null},{"url":"/paper/fuxi-a-benchmark-for-evaluating-language","slug":"fuxi-a-benchmark-for-evaluating-language","title":"Fùxì: A Benchmark for Evaluating Language Models on Ancient Chinese Text Understanding and Generation","date":"2025-03-20","arxiv_id":"2503.15837","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-level-instruction-injection-for-video","slug":"hybrid-level-instruction-injection-for-video","title":"Hybrid-Level Instruction Injection for Video Token Compression in Multi-modal Large Language Models","date":"2025-03-20","arxiv_id":"2503.16036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-level-instruction-injection-for-video#ran","syntology_url":"https://syntology.ai/paper/2503.16036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16036"}},"official":{"repos":["lntzm/hicom"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-much-do-llms-learn-from-negative-examples","slug":"how-much-do-llms-learn-from-negative-examples","title":"How much do LLMs learn from negative examples?","date":"2025-03-18","arxiv_id":"2503.14391","repositories_listed":1,"syntology":null},{"url":"/paper/leavs-an-llm-based-labeler-for-abdominal-ct","slug":"leavs-an-llm-based-labeler-for-abdominal-ct","title":"LEAVS: An LLM-based Labeler for Abdominal CT Supervision","date":"2025-03-17","arxiv_id":"2503.13330","repositories_listed":1,"syntology":null},{"url":"/paper/microvqa-a-multimodal-reasoning-benchmark-for","slug":"microvqa-a-multimodal-reasoning-benchmark-for","title":"MicroVQA: A Multimodal Reasoning Benchmark for Microscopy-Based Scientific Research","date":"2025-03-17","arxiv_id":"2503.13399","repositories_listed":1,"syntology":null},{"url":"/paper/seqsam-autoregressive-multiple-hypothesis","slug":"seqsam-autoregressive-multiple-hypothesis","title":"SeqSAM: Autoregressive Multiple Hypothesis Prediction for Medical Image Segmentation using SAM","date":"2025-03-12","arxiv_id":"2503.09797","repositories_listed":1,"syntology":null},{"url":"/paper/mellow-a-small-audio-language-model-for","slug":"mellow-a-small-audio-language-model-for","title":"Mellow: a small audio language model for reasoning","date":"2025-03-11","arxiv_id":"2503.08540","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mellow-a-small-audio-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2503.08540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08540"}},"official":{"repos":["soham97/mellow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/visbias-measuring-explicit-and-implicit","slug":"visbias-measuring-explicit-and-implicit","title":"VisBias: Measuring Explicit and Implicit Social Biases in Vision Language Models","date":"2025-03-10","arxiv_id":"2503.07575","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visbias-measuring-explicit-and-implicit#ran","syntology_url":"https://syntology.ai/paper/2503.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07575"}},"official":{"repos":["uscnlp-lime/visbias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cupcase-clinically-uncommon-patient-cases-and","slug":"cupcase-clinically-uncommon-patient-cases-and","title":"CUPCase: Clinically Uncommon Patient Cases and Diagnoses Dataset","date":"2025-03-08","arxiv_id":"2503.06204","repositories_listed":1,"syntology":null},{"url":"/paper/knowlogic-a-benchmark-for-commonsense","slug":"knowlogic-a-benchmark-for-commonsense","title":"SCoRE: Benchmarking Long-Chain Reasoning in Commonsense Scenarios","date":"2025-03-08","arxiv_id":"2503.06218","repositories_listed":1,"syntology":null},{"url":"/paper/this-is-your-doge-if-it-please-you-exploring","slug":"this-is-your-doge-if-it-please-you-exploring","title":"This Is Your Doge, If It Please You: Exploring Deception and Robustness in Mixture of LLMs","date":"2025-03-07","arxiv_id":"2503.05856","repositories_listed":1,"syntology":null},{"url":"/paper/analogical-reasoning-inside-large-language","slug":"analogical-reasoning-inside-large-language","title":"Analogical Reasoning Inside Large Language Models: Concept Vectors and the Limits of Abstraction","date":"2025-03-05","arxiv_id":"2503.03666","repositories_listed":1,"syntology":null},{"url":"/paper/when-an-llm-is-apprehensive-about-its-answers","slug":"when-an-llm-is-apprehensive-about-its-answers","title":"When an LLM is apprehensive about its answers -- and when its uncertainty is justified","date":"2025-03-03","arxiv_id":"2503.01688","repositories_listed":1,"syntology":null},{"url":"/paper/eaira-establishing-a-methodology-for","slug":"eaira-establishing-a-methodology-for","title":"EAIRA: Establishing a Methodology for Evaluating AI Models as Scientific Research Assistants","date":"2025-02-27","arxiv_id":"2502.20309","repositories_listed":1,"syntology":null},{"url":"/paper/wicked-a-simple-method-to-make-multiple","slug":"wicked-a-simple-method-to-make-multiple","title":"WiCkeD: A Simple Method to Make Multiple Choice Benchmarks More Challenging","date":"2025-02-25","arxiv_id":"2502.18316","repositories_listed":1,"syntology":null},{"url":"/paper/autologi-automated-generation-of-logic","slug":"autologi-automated-generation-of-logic","title":"AutoLogi: Automated Generation of Logic Puzzles for Evaluating Reasoning Abilities of Large Language Models","date":"2025-02-24","arxiv_id":"2502.16906","repositories_listed":1,"syntology":null},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/moving-beyond-medical-exam-questions-a","slug":"moving-beyond-medical-exam-questions-a","title":"Moving Beyond Medical Exam Questions: A Clinician-Annotated Dataset of Real-World Tasks and Ambiguity in Mental Healthcare","date":"2025-02-22","arxiv_id":"2502.16051","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/moving-beyond-medical-exam-questions-a#ran","syntology_url":"https://syntology.ai/paper/2502.16051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16051"}},"official":{"repos":["maxlampe/mentat"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wrong-answers-can-also-be-useful-plausibleqa","slug":"wrong-answers-can-also-be-useful-plausibleqa","title":"Wrong Answers Can Also Be Useful: PlausibleQA -- A Large-Scale QA Dataset with Answer Plausibility Scores","date":"2025-02-22","arxiv_id":"2502.16358","repositories_listed":1,"syntology":null},{"url":"/paper/mind-the-confidence-gap-overconfidence","slug":"mind-the-confidence-gap-overconfidence","title":"Mind the Confidence Gap: Overconfidence, Calibration, and Distractor Effects in Large Language Models","date":"2025-02-16","arxiv_id":"2502.11028","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-confidence-gap-overconfidence#ran","syntology_url":"https://syntology.ai/paper/2502.11028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11028"}},"official":{"repos":["prateekchhikara/llms-calibration"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/truth-knows-no-language-evaluating","slug":"truth-knows-no-language-evaluating","title":"Truth Knows No Language: Evaluating Truthfulness Beyond English","date":"2025-02-13","arxiv_id":"2502.09387","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/truth-knows-no-language-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.09387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09387"}},"official":{"repos":["hitz-zentroa/truthfulqa-multi"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/let-the-ai-conspiracy-begin-language-model","slug":"let-the-ai-conspiracy-begin-language-model","title":"HSI: Head-Specific Intervention Can Induce Misaligned AI Coordination in Large Language Models","date":"2025-02-09","arxiv_id":"2502.05945","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-shortcomings-of-llms-in","slug":"investigating-the-shortcomings-of-llms-in","title":"Investigating the Shortcomings of LLMs in Step-by-Step Legal Reasoning","date":"2025-02-08","arxiv_id":"2502.05675","repositories_listed":1,"syntology":null},{"url":"/paper/arr-question-answering-with-large-language","slug":"arr-question-answering-with-large-language","title":"ARR: Question Answering with Large Language Models via Analyzing, Retrieving, and Reasoning","date":"2025-02-07","arxiv_id":"2502.04689","repositories_listed":1,"syntology":null},{"url":"/paper/tumtraffic-videoqa-a-benchmark-for-unified","slug":"tumtraffic-videoqa-a-benchmark-for-unified","title":"TUMTraffic-VideoQA: A Benchmark for Unified Spatio-Temporal Video Understanding in Traffic Scenes","date":"2025-02-04","arxiv_id":"2502.02449","repositories_listed":1,"syntology":null},{"url":"/paper/option-id-based-elimination-for-multiple","slug":"option-id-based-elimination-for-multiple","title":"Option-ID Based Elimination For Multiple Choice Questions","date":"2025-01-25","arxiv_id":"2501.15175","repositories_listed":1,"syntology":null},{"url":"/paper/patent-figure-classification-using-large","slug":"patent-figure-classification-using-large","title":"Patent Figure Classification using Large Vision-language Models","date":"2025-01-22","arxiv_id":"2501.12751","repositories_listed":1,"syntology":null},{"url":"/paper/meds-3-towards-medical-small-language-models","slug":"meds-3-towards-medical-small-language-models","title":"MedS$^3$: Towards Medical Small Language Models with Self-Evolved Slow Thinking","date":"2025-01-21","arxiv_id":"2501.12051","repositories_listed":1,"syntology":null},{"url":"/paper/facexbench-evaluating-multimodal-llms-on-face","slug":"facexbench-evaluating-multimodal-llms-on-face","title":"FaceXBench: Evaluating Multimodal LLMs on Face Understanding","date":"2025-01-17","arxiv_id":"2501.10360","repositories_listed":1,"syntology":null},{"url":"/paper/tomato-verbalizing-the-mental-states-of-role","slug":"tomato-verbalizing-the-mental-states-of-role","title":"ToMATO: Verbalizing the Mental States of Role-Playing LLMs for Benchmarking Theory of Mind","date":"2025-01-15","arxiv_id":"2501.08838","repositories_listed":1,"syntology":null},{"url":"/paper/zno-eval-benchmarking-reasoning-capabilities","slug":"zno-eval-benchmarking-reasoning-capabilities","title":"ZNO-Eval: Benchmarking reasoning capabilities of large language models in Ukrainian","date":"2025-01-12","arxiv_id":"2501.06715","repositories_listed":1,"syntology":null},{"url":"/paper/affordably-fine-tuned-llms-provide-better","slug":"affordably-fine-tuned-llms-provide-better","title":"Affordably Fine-tuned LLMs Provide Better Answers to Course-specific MCQs","date":"2025-01-10","arxiv_id":"2501.05891","repositories_listed":1,"syntology":null},{"url":"/paper/fleurs-slu-a-massively-multilingual-benchmark","slug":"fleurs-slu-a-massively-multilingual-benchmark","title":"Fleurs-SLU: A Massively Multilingual Benchmark for Spoken Language Understanding","date":"2025-01-10","arxiv_id":"2501.06117","repositories_listed":1,"syntology":null},{"url":"/paper/automated-generation-of-challenging-multiple","slug":"automated-generation-of-challenging-multiple","title":"Automated Generation of Challenging Multiple-Choice Questions for Vision Language Model Evaluation","date":"2025-01-06","arxiv_id":"2501.03225","repositories_listed":1,"syntology":null},{"url":"/paper/whyphi-fine-tuning-phi-3-for-multiple-choice","slug":"whyphi-fine-tuning-phi-3-for-multiple-choice","title":"(WhyPHI) Fine-Tuning PHI-3 for Multiple-Choice Question Answering: Methodology, Results, and Challenges","date":"2025-01-03","arxiv_id":"2501.01588","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-specialized-visual-encoders-for","slug":"unifying-specialized-visual-encoders-for","title":"Unifying Specialized Visual Encoders for Video Language Models","date":"2025-01-02","arxiv_id":"2501.01426","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-specialized-visual-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2501.01426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01426"}},"official":{"repos":["princetonvisualai/merv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longbench-v2-towards-deeper-understanding-and","slug":"longbench-v2-towards-deeper-understanding-and","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","date":"2024-12-19","arxiv_id":"2412.15204","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longbench-v2-towards-deeper-understanding-and#ran","syntology_url":"https://syntology.ai/paper/2412.15204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15204"}},"official":{"repos":["thudm/longbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlu-cf-a-contamination-free-multi-task","slug":"mmlu-cf-a-contamination-free-multi-task","title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","date":"2024-12-19","arxiv_id":"2412.15194","repositories_listed":1,"syntology":null},{"url":"/paper/medg-krp-medical-graph-knowledge","slug":"medg-krp-medical-graph-knowledge","title":"MedG-KRP: Medical Graph Knowledge Representation Probing","date":"2024-12-14","arxiv_id":"2412.10982","repositories_listed":1,"syntology":null},{"url":"/paper/does-multiple-choice-have-a-future-in-the-age","slug":"does-multiple-choice-have-a-future-in-the-age","title":"Does Multiple Choice Have a Future in the Age of Generative AI? A Posttest-only RCT","date":"2024-12-13","arxiv_id":"2412.10267","repositories_listed":1,"syntology":null},{"url":"/paper/improve-impact-of-mobile-phones-on-remote","slug":"improve-impact-of-mobile-phones-on-remote","title":"A multimodal dataset for understanding the impact of mobile phones on remote online virtual education","date":"2024-12-13","arxiv_id":"2412.14195","repositories_listed":1,"syntology":null},{"url":"/paper/filter-then-generate-large-language-models","slug":"filter-then-generate-large-language-models","title":"Filter-then-Generate: Large Language Models with Structure-Text Adapter for Knowledge Graph Completion","date":"2024-12-12","arxiv_id":"2412.09094","repositories_listed":1,"syntology":null},{"url":"/paper/neptune-the-long-orbit-to-benchmarking-long","slug":"neptune-the-long-orbit-to-benchmarking-long","title":"Neptune: The Long Orbit to Benchmarking Long Video Understanding","date":"2024-12-12","arxiv_id":"2412.09582","repositories_listed":1,"syntology":null},{"url":"/paper/mm-poe-multiple-choice-reasoning-via-process","slug":"mm-poe-multiple-choice-reasoning-via-process","title":"MM-PoE: Multiple Choice Reasoning via. Process of Elimination using Multi-Modal Models","date":"2024-12-10","arxiv_id":"2412.07148","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-and-mitigating-social-bias-for","slug":"evaluating-and-mitigating-social-bias-for","title":"Evaluating and Mitigating Social Bias for Large Language Models in Open-ended Settings","date":"2024-12-09","arxiv_id":"2412.06134","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-and-mitigating-social-bias-for#ran","syntology_url":"https://syntology.ai/paper/2412.06134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06134"}},"official":{"repos":["zhaoliu0914/LLM-Bias-Benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-correction-explainable-feedback","slug":"learning-to-correction-explainable-feedback","title":"Learning to Correction: Explainable Feedback Generation for Visual Commonsense Reasoning Distractor","date":"2024-12-08","arxiv_id":"2412.07801","repositories_listed":1,"syntology":null},{"url":"/paper/av-odyssey-bench-can-your-multimodal-llms","slug":"av-odyssey-bench-can-your-multimodal-llms","title":"AV-Odyssey Bench: Can Your Multimodal LLMs Really Understand Audio-Visual Information?","date":"2024-12-03","arxiv_id":"2412.02611","repositories_listed":1,"syntology":null},{"url":"/paper/noise-injection-reveals-hidden-capabilities","slug":"noise-injection-reveals-hidden-capabilities","title":"Noise Injection Reveals Hidden Capabilities of Sandbagging Language Models","date":"2024-12-02","arxiv_id":"2412.01784","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-injection-reveals-hidden-capabilities#ran","syntology_url":"https://syntology.ai/paper/2412.01784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01784"}},"official":{"repos":["camtice/sandbagdetect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/sailcompass-towards-reproducible-and-robust","slug":"sailcompass-towards-reproducible-and-robust","title":"SailCompass: Towards Reproducible and Robust Evaluation for Southeast Asian Languages","date":"2024-12-02","arxiv_id":"2412.01186","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-abilities-of-large-language","slug":"exploring-the-abilities-of-large-language","title":"KnowledgePrompts: Exploring the Abilities of Large Language Models to Solve Proportional Analogies via Knowledge-Enhanced Prompting","date":"2024-12-01","arxiv_id":"2412.00869","repositories_listed":1,"syntology":null},{"url":"/paper/visonlyqa-large-vision-language-models-still","slug":"visonlyqa-large-vision-language-models-still","title":"VisOnlyQA: Large Vision Language Models Still Struggle with Visual Perception of Geometric Information","date":"2024-12-01","arxiv_id":"2412.00947","repositories_listed":1,"syntology":null},{"url":"/paper/coreval-a-comprehensive-and-objective","slug":"coreval-a-comprehensive-and-objective","title":"CHOICE: Benchmarking the Remote Sensing Capabilities of Large Vision-Language Models","date":"2024-11-27","arxiv_id":"2411.18145","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/coreval-a-comprehensive-and-objective#ran","syntology_url":"https://syntology.ai/paper/2411.18145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18145"}},"official":{"repos":["shawnan-whu/choice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/all-languages-matter-evaluating-lmms-on","slug":"all-languages-matter-evaluating-lmms-on","title":"All Languages Matter: Evaluating LMMs on Culturally Diverse 100 Languages","date":"2024-11-25","arxiv_id":"2411.16508","repositories_listed":1,"syntology":null},{"url":"/paper/dahl-domain-specific-automated-hallucination","slug":"dahl-domain-specific-automated-hallucination","title":"DAHL: Domain-specific Automated Hallucination Evaluation of Long-Form Text through a Benchmark Dataset in Biomedicine","date":"2024-11-14","arxiv_id":"2411.09255","repositories_listed":1,"syntology":null},{"url":"/paper/trace-transformer-based-risk-assessment-for","slug":"trace-transformer-based-risk-assessment-for","title":"TRACE: Transformer-based Risk Assessment for Clinical Evaluation","date":"2024-11-13","arxiv_id":"2411.08701","repositories_listed":1,"syntology":null},{"url":"/paper/identifyme-a-challenging-long-context-mention","slug":"identifyme-a-challenging-long-context-mention","title":"IdentifyMe: A Challenging Long-Context Mention Resolution Benchmark for LLMs","date":"2024-11-12","arxiv_id":"2411.07466","repositories_listed":1,"syntology":null},{"url":"/paper/storyteller-improving-long-video-description","slug":"storyteller-improving-long-video-description","title":"StoryTeller: Improving Long Video Description through Global Audio-Visual Character Identification","date":"2024-11-11","arxiv_id":"2411.07076","repositories_listed":1,"syntology":null},{"url":"/paper/quantitative-assessment-of-intersectional","slug":"quantitative-assessment-of-intersectional","title":"Quantitative Assessment of Intersectional Empathetic Bias and Understanding","date":"2024-11-08","arxiv_id":"2411.05777","repositories_listed":1,"syntology":null},{"url":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hourvideo-1-hour-video-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.04998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04998"}},"official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/meg-medical-knowledge-augmented-large","slug":"meg-medical-knowledge-augmented-large","title":"MEG: Medical Knowledge-Augmented Large Language Models for Question Answering","date":"2024-11-06","arxiv_id":"2411.03883","repositories_listed":1,"syntology":null},{"url":"/paper/milu-a-multi-task-indic-language","slug":"milu-a-multi-task-indic-language","title":"MILU: A Multi-task Indic Language Understanding Benchmark","date":"2024-11-04","arxiv_id":"2411.02538","repositories_listed":1,"syntology":null},{"url":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ppllava-varied-video-sequence-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.02327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02327"}},"official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/improving-model-evaluation-using-smart","slug":"improving-model-evaluation-using-smart","title":"Improving Model Evaluation using SMART Filtering of Benchmark Datasets","date":"2024-10-26","arxiv_id":"2410.20245","repositories_listed":1,"syntology":null},{"url":"/paper/delving-into-the-reversal-curse-how-far-can","slug":"delving-into-the-reversal-curse-how-far-can","title":"Delving into the Reversal Curse: How Far Can Large Language Models Generalize?","date":"2024-10-24","arxiv_id":"2410.18808","repositories_listed":1,"syntology":null},{"url":"/paper/how-can-we-diagnose-and-treat-bias-in-large","slug":"how-can-we-diagnose-and-treat-bias-in-large","title":"How Can We Diagnose and Treat Bias in Large Language Models for Clinical Decision-Making?","date":"2024-10-21","arxiv_id":"2410.16574","repositories_listed":1,"syntology":null},{"url":"/paper/timeseriesexam-a-time-series-understanding","slug":"timeseriesexam-a-time-series-understanding","title":"TimeSeriesExam: A time series understanding exam","date":"2024-10-18","arxiv_id":"2410.14752","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timeseriesexam-a-time-series-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.14752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14752"}},"official":null}},{"url":"/paper/mcqg-srefine-multiple-choice-question","slug":"mcqg-srefine-multiple-choice-question","title":"MCQG-SRefine: Multiple Choice Question Generation and Evaluation with Iterative Self-Critique, Correction, and Comparison Feedback","date":"2024-10-17","arxiv_id":"2410.13191","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-instruction-following","slug":"evaluating-the-instruction-following","title":"Evaluating the Instruction-following Abilities of Language Models using Knowledge Tasks","date":"2024-10-16","arxiv_id":"2410.12972","repositories_listed":1,"syntology":null},{"url":"/paper/worldmedqa-v-a-multilingual-multimodal","slug":"worldmedqa-v-a-multilingual-multimodal","title":"WorldMedQA-V: a multilingual, multimodal medical examination dataset for multimodal language models evaluation","date":"2024-10-16","arxiv_id":"2410.12722","repositories_listed":1,"syntology":null},{"url":"/paper/difficult-task-yes-but-simple-task-no","slug":"difficult-task-yes-but-simple-task-no","title":"Difficult Task Yes but Simple Task No: Unveiling the Laziness in Multimodal LLMs","date":"2024-10-15","arxiv_id":"2410.11437","repositories_listed":1,"syntology":null},{"url":"/paper/leaving-the-barn-door-open-for-clever-hans","slug":"leaving-the-barn-door-open-for-clever-hans","title":"Leaving the barn door open for Clever Hans: Simple features predict LLM benchmark answers","date":"2024-10-15","arxiv_id":"2410.11672","repositories_listed":1,"syntology":null},{"url":"/paper/mmie-massive-multimodal-interleaved","slug":"mmie-massive-multimodal-interleaved","title":"MMIE: Massive Multimodal Interleaved Comprehension Benchmark for Large Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.10139","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmie-massive-multimodal-interleaved#ran","syntology_url":"https://syntology.ai/paper/2410.10139","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10139"}},"official":{"repos":["Lillianwei-h/MMIE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/longhalqa-long-context-hallucination","slug":"longhalqa-long-context-hallucination","title":"LongHalQA: Long-Context Hallucination Evaluation for MultiModal Large Language Models","date":"2024-10-13","arxiv_id":"2410.09962","repositories_listed":1,"syntology":null},{"url":"/paper/taming-overconfidence-in-llms-reward","slug":"taming-overconfidence-in-llms-reward","title":"Taming Overconfidence in LLMs: Reward Calibration in RLHF","date":"2024-10-13","arxiv_id":"2410.09724","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/taming-overconfidence-in-llms-reward#ran","syntology_url":"https://syntology.ai/paper/2410.09724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09724"}},"official":{"repos":["SeanLeng1/Reward-Calibration"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/novo-norm-voting-off-hallucinations-with","slug":"novo-norm-voting-off-hallucinations-with","title":"NoVo: Norm Voting off Hallucinations with Attention Heads in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08970","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/novo-norm-voting-off-hallucinations-with#ran","syntology_url":"https://syntology.ai/paper/2410.08970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08970"}},"official":null}},{"url":"/paper/sportu-a-comprehensive-sports-understanding","slug":"sportu-a-comprehensive-sports-understanding","title":"SPORTU: A Comprehensive Sports Understanding Benchmark for Multimodal Large Language Models","date":"2024-10-11","arxiv_id":"2410.08474","repositories_listed":1,"syntology":null},{"url":"/paper/utilize-the-flow-before-stepping-into-the","slug":"utilize-the-flow-before-stepping-into-the","title":"Utilize the Flow before Stepping into the Same River Twice: Certainty Represented Knowledge Flow for Refusal-Aware Instruction Tuning","date":"2024-10-09","arxiv_id":"2410.06913","repositories_listed":1,"syntology":null},{"url":"/paper/plausibly-problematic-questions-in-multiple","slug":"plausibly-problematic-questions-in-multiple","title":"Plausibly Problematic Questions in Multiple-Choice Benchmarks for Commonsense Reasoning","date":"2024-10-06","arxiv_id":"2410.10854","repositories_listed":1,"syntology":null},{"url":"/paper/dlp-lora-efficient-task-specific-lora-fusion","slug":"dlp-lora-efficient-task-specific-lora-fusion","title":"DLP-LoRA: Efficient Task-Specific LoRA Fusion with a Dynamic, Lightweight Plugin for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01497","repositories_listed":1,"syntology":null},{"url":"/paper/introducing-flexible-monotone-multiple-choice","slug":"introducing-flexible-monotone-multiple-choice","title":"Introducing Flexible Monotone Multiple Choice Item Response Theory Models and Bit Scales","date":"2024-10-02","arxiv_id":"2410.01480","repositories_listed":1,"syntology":null},{"url":"/paper/medqa-cs-benchmarking-large-language-models","slug":"medqa-cs-benchmarking-large-language-models","title":"MedQA-CS: Benchmarking Large Language Models Clinical Skills Using an AI-SCE Framework","date":"2024-10-02","arxiv_id":"2410.01553","repositories_listed":1,"syntology":null},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/disgem-distractor-generation-for-multiple","slug":"disgem-distractor-generation-for-multiple","title":"DisGeM: Distractor Generation for Multiple Choice Questions with Span Masking","date":"2024-09-26","arxiv_id":"2409.18263","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-healthcare-llms-through-retrieved","slug":"boosting-healthcare-llms-through-retrieved","title":"Boosting Healthcare LLMs Through Retrieved Context","date":"2024-09-23","arxiv_id":"2409.15127","repositories_listed":1,"syntology":null},{"url":"/paper/2409-14175","slug":"2409-14175","title":"QMOS: Enhancing LLMs for Telecommunication with Question Masked loss and Option Shuffling","date":"2024-09-21","arxiv_id":"2409.14175","repositories_listed":1,"syntology":null},{"url":"/paper/annealed-winner-takes-all-for-motion","slug":"annealed-winner-takes-all-for-motion","title":"Annealed Winner-Takes-All for Motion Forecasting","date":"2024-09-17","arxiv_id":"2409.11172","repositories_listed":1,"syntology":null},{"url":"/paper/columbus-evaluating-cognitive-lateral","slug":"columbus-evaluating-cognitive-lateral","title":"COLUMBUS: Evaluating COgnitive Lateral Understanding through Multiple-choice reBUSes","date":"2024-09-06","arxiv_id":"2409.04053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/columbus-evaluating-cognitive-lateral#ran","syntology_url":"https://syntology.ai/paper/2409.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04053"}},"official":{"repos":["koen-47/columbus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/training-on-the-benchmark-is-not-all-you-need","slug":"training-on-the-benchmark-is-not-all-you-need","title":"Training on the Benchmark Is Not All You Need","date":"2024-09-03","arxiv_id":"2409.01790","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-on-the-benchmark-is-not-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2409.01790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01790"}},"official":{"repos":["nishiwen1214/Benchmark-leakage-detection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toursynbio-a-multi-modal-large-model-and","slug":"toursynbio-a-multi-modal-large-model-and","title":"TourSynbio: A Multi-Modal Large Model and Agent Framework to Bridge Text and Protein Sequences for Protein Engineering","date":"2024-08-27","arxiv_id":"2408.15299","repositories_listed":1,"syntology":null},{"url":"/paper/wait-that-s-not-an-option-llms-robustness","slug":"wait-that-s-not-an-option-llms-robustness","title":"Wait, that's not an option: LLMs Robustness with Incorrect Multiple-Choice Options","date":"2024-08-27","arxiv_id":"2409.00113","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wait-that-s-not-an-option-llms-robustness#ran","syntology_url":"https://syntology.ai/paper/2409.00113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00113"}},"official":{"repos":["gracjangoral/when-all-options-are-wrong"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-and-large-language-model","slug":"vision-language-and-large-language-model","title":"Vision-Language and Large Language Model Performance in Gastroenterology: GPT, Claude, Llama, Phi, Mistral, Gemma, and Quantized Models","date":"2024-08-25","arxiv_id":"2409.00084","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-knowledge-tracing-with-concept-map","slug":"enhancing-knowledge-tracing-with-concept-map","title":"Enhancing Knowledge Tracing with Concept Map and Response Disentanglement","date":"2024-08-23","arxiv_id":"2408.12996","repositories_listed":1,"syntology":null},{"url":"/paper/towards-evaluating-and-building-versatile","slug":"towards-evaluating-and-building-versatile","title":"Towards Evaluating and Building Versatile Large Language Models for Medicine","date":"2024-08-22","arxiv_id":"2408.12547","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-evaluating-and-building-versatile#ran","syntology_url":"https://syntology.ai/paper/2408.12547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12547"}},"official":{"repos":["magic-ai4med/meds-ins"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiating-choices-via-commonality-for","slug":"differentiating-choices-via-commonality-for","title":"Differentiating Choices via Commonality for Multiple-Choice Question Answering","date":"2024-08-21","arxiv_id":"2408.11554","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-visual-sycophancy-in-multimodal","slug":"measuring-visual-sycophancy-in-multimodal","title":"Measuring Agreeableness Bias in Multimodal Models","date":"2024-08-17","arxiv_id":"2408.09111","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-visual-sycophancy-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.09111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09111"}},"official":{"repos":["jasonlim131/looksRdeceiving"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-exemplar-enhancing-distractor","slug":"chain-of-exemplar-enhancing-distractor","title":"Chain-of-Exemplar: Enhancing Distractor Generation for Multimodal Educational Question Generation","date":"2024-08-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/llms-are-biased-towards-output-formats","slug":"llms-are-biased-towards-output-formats","title":"LLMs Are Biased Towards Output Formats! Systematically Evaluating and Mitigating Output Format Bias of LLMs","date":"2024-08-16","arxiv_id":"2408.08656","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llms-are-biased-towards-output-formats#ran","syntology_url":"https://syntology.ai/paper/2408.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08656"}},"official":{"repos":["dxlong2000/FormatEval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"976f548f385afee4ccebd24aef056f800c245a8206f8db005866cf640145d2c2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}