{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/3","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":12,"rows_per_page":100,"rows":[201,300],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/2","next":"/task/multiple-choice/papers/4","papers":[{"url":"/paper/2408-02718","slug":"2408-02718","title":"MMIU: Multimodal Multi-image Understanding for Evaluating Large Vision-Language Models","date":"2024-08-05","arxiv_id":"2408.02718","repositories_listed":1,"syntology":null},{"url":"/paper/xmainframe-a-large-language-model-for","slug":"xmainframe-a-large-language-model-for","title":"XMainframe: A Large Language Model for Mainframe Modernization","date":"2024-08-05","arxiv_id":"2408.04660","repositories_listed":1,"syntology":null},{"url":"/paper/annealed-multiple-choice-learning-overcoming","slug":"annealed-multiple-choice-learning-overcoming","title":"Annealed Multiple Choice Learning: Overcoming limitations of Winner-takes-all with annealing","date":"2024-07-22","arxiv_id":"2407.15580","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/annealed-multiple-choice-learning-overcoming#ran","syntology_url":"https://syntology.ai/paper/2407.15580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15580"}},"official":{"repos":["victorletzelter/annealed_mcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}},"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modular-sentence-encoders-separating-language","slug":"modular-sentence-encoders-separating-language","title":"Modular Sentence Encoders: Separating Language Specialization from Cross-Lingual Alignment","date":"2024-07-20","arxiv_id":"2407.14878","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-language-models-as-risk-scores","slug":"evaluating-language-models-as-risk-scores","title":"Evaluating language models as risk scores","date":"2024-07-19","arxiv_id":"2407.14614","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/evaluating-language-models-as-risk-scores#ran","syntology_url":"https://syntology.ai/paper/2407.14614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14614"}},"official":{"repos":["socialfoundations/folktexts"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/turkishmmlu-measuring-massive-multitask","slug":"turkishmmlu-measuring-massive-multitask","title":"TurkishMMLU: Measuring Massive Multitask Language Understanding in Turkish","date":"2024-07-17","arxiv_id":"2407.12402","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-large-language-models-for","slug":"harnessing-large-language-models-for","title":"Fine-tuning Multimodal Large Language Models for Product Bundling","date":"2024-07-16","arxiv_id":"2407.11712","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2407.11712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11712"}},"official":{"repos":["xiaohao-liu/bundle-mllm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-large-language-models-for-nano","slug":"leveraging-large-language-models-for-nano","title":"Leveraging large language models for nano synthesis mechanism explanation: solid foundations or mere conjectures?","date":"2024-07-12","arxiv_id":"2407.08922","repositories_listed":1,"syntology":null},{"url":"/paper/self-recognition-in-language-models","slug":"self-recognition-in-language-models","title":"Self-Recognition in Language Models","date":"2024-07-09","arxiv_id":"2407.06946","repositories_listed":1,"syntology":null},{"url":"/paper/oran-bench-13k-an-open-source-benchmark-for","slug":"oran-bench-13k-an-open-source-benchmark-for","title":"ORAN-Bench-13K: An Open Source Benchmark for Assessing LLMs in Open Radio Access Networks","date":"2024-07-08","arxiv_id":"2407.06245","repositories_listed":1,"syntology":null},{"url":"/paper/can-model-uncertainty-function-as-a-proxy-for","slug":"can-model-uncertainty-function-as-a-proxy-for","title":"Can Model Uncertainty Function as a Proxy for Multiple-Choice Question Item Difficulty?","date":"2024-07-07","arxiv_id":"2407.05327","repositories_listed":1,"syntology":null},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mmsci-a-multimodal-multi-discipline-dataset","slug":"mmsci-a-multimodal-multi-discipline-dataset","title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","date":"2024-07-06","arxiv_id":"2407.04903","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mmsci-a-multimodal-multi-discipline-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.04903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04903"}},"official":{"repos":["leezekun/mmsci"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-consistent-over","slug":"are-large-language-models-consistent-over","title":"Are Large Language Models Consistent over Value-laden Questions?","date":"2024-07-03","arxiv_id":"2407.02996","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-large-language-models-consistent-over#ran","syntology_url":"https://syntology.ai/paper/2407.02996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02996"}},"official":{"repos":["jlcmoore/ValueConsistency"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-your-large-language-model-knowledgeable-or","slug":"is-your-large-language-model-knowledgeable-or","title":"Is Your Large Language Model Knowledgeable or a Choices-Only Cheater?","date":"2024-07-02","arxiv_id":"2407.01992","repositories_listed":1,"syntology":null},{"url":"/paper/mmevalpro-calibrating-multimodal-benchmarks","slug":"mmevalpro-calibrating-multimodal-benchmarks","title":"MMEvalPro: Calibrating Multimodal Benchmarks Towards Trustworthy and Efficient Evaluation","date":"2024-06-29","arxiv_id":"2407.00468","repositories_listed":1,"syntology":null},{"url":"/paper/infinibench-a-comprehensive-benchmark-for","slug":"infinibench-a-comprehensive-benchmark-for","title":"InfiniBench: A Comprehensive Benchmark for Large Multimodal Models in Very Long Video Understanding","date":"2024-06-28","arxiv_id":"2406.19875","repositories_listed":1,"syntology":null},{"url":"/paper/divert-distractor-generation-with-variational","slug":"divert-distractor-generation-with-variational","title":"DiVERT: Distractor Generation with Variational Errors Represented as Text for Math Multiple-choice Questions","date":"2024-06-27","arxiv_id":"2406.19356","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/divert-distractor-generation-with-variational#ran","syntology_url":"https://syntology.ai/paper/2406.19356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19356"}},"official":{"repos":["umass-ml4ed/divert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/length-optimization-in-conformal-prediction","slug":"length-optimization-in-conformal-prediction","title":"Length Optimization in Conformal Prediction","date":"2024-06-27","arxiv_id":"2406.18814","repositories_listed":1,"syntology":null},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hcqa-ego4d-egoschema-challenge-2024","slug":"hcqa-ego4d-egoschema-challenge-2024","title":"HCQA @ Ego4D EgoSchema Challenge 2024","date":"2024-06-22","arxiv_id":"2406.15771","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hcqa-ego4d-egoschema-challenge-2024#ran","syntology_url":"https://syntology.ai/paper/2406.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15771"}},"official":{"repos":["hyu-zhang/hcqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/african-or-european-swallow-benchmarking","slug":"african-or-european-swallow-benchmarking","title":"African or European Swallow? Benchmarking Large Vision-Language Models for Fine-Grained Object Classification","date":"2024-06-20","arxiv_id":"2406.14496","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/african-or-european-swallow-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.14496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14496"}},"official":{"repos":["gregor-ge/foci-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clinicallab-aligning-agents-for-multi","slug":"clinicallab-aligning-agents-for-multi","title":"ClinicalLab: Aligning Agents for Multi-Departmental Clinical Diagnostics in the Real World","date":"2024-06-19","arxiv_id":"2406.13890","repositories_listed":1,"syntology":null},{"url":"/paper/detectbench-can-large-language-model-detect","slug":"detectbench-can-large-language-model-detect","title":"DetectBench: Can Large Language Model Detect and Piece Together Implicit Evidence?","date":"2024-06-18","arxiv_id":"2406.12641","repositories_listed":1,"syntology":null},{"url":"/paper/ipeval-a-bilingual-intellectual-property","slug":"ipeval-a-bilingual-intellectual-property","title":"IPEval: A Bilingual Intellectual Property Agency Consultation Evaluation Benchmark for Large Language Models","date":"2024-06-18","arxiv_id":"2406.12386","repositories_listed":1,"syntology":null},{"url":"/paper/ubench-benchmarking-uncertainty-in-large","slug":"ubench-benchmarking-uncertainty-in-large","title":"UBENCH: Benchmarking Uncertainty in Large Language Models with Multiple Choice Questions","date":"2024-06-18","arxiv_id":"2406.12784","repositories_listed":1,"syntology":null},{"url":"/paper/grade-score-quantifying-llm-performance-in","slug":"grade-score-quantifying-llm-performance-in","title":"Grade Score: Quantifying LLM Performance in Option Selection","date":"2024-06-17","arxiv_id":"2406.12043","repositories_listed":1,"syntology":null},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/not-all-bias-is-bad-balancing-rational","slug":"not-all-bias-is-bad-balancing-rational","title":"Balancing Rigor and Utility: Mitigating Cognitive Biases in Large Language Models for Multiple-Choice Questions","date":"2024-06-16","arxiv_id":"2406.10999","repositories_listed":1,"syntology":null},{"url":"/paper/color-filter-conditional-loss-reduction","slug":"color-filter-conditional-loss-reduction","title":"CoLoR-Filter: Conditional Loss Reduction Filtering for Targeted Language Model Pre-training","date":"2024-06-15","arxiv_id":"2406.10670","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/color-filter-conditional-loss-reduction#ran","syntology_url":"https://syntology.ai/paper/2406.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10670"}},"official":{"repos":["davidbrandfonbrener/color-filter-olmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blend-a-benchmark-for-llms-on-everyday","slug":"blend-a-benchmark-for-llms-on-everyday","title":"BLEnD: A Benchmark for LLMs on Everyday Knowledge in Diverse Cultures and Languages","date":"2024-06-14","arxiv_id":"2406.09948","repositories_listed":1,"syntology":{"n":10,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":10,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/blend-a-benchmark-for-llms-on-everyday#ran","syntology_url":"https://syntology.ai/paper/2406.09948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09948"}},"official":{"repos":["nlee0212/blend"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/chisafetybench-a-chinese-hierarchical-safety","slug":"chisafetybench-a-chinese-hierarchical-safety","title":"CHiSafetyBench: A Chinese Hierarchical Safety Benchmark for Large Language Models","date":"2024-06-14","arxiv_id":"2406.10311","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-chatgpt-4-vision-on-brazil-s","slug":"evaluating-chatgpt-4-vision-on-brazil-s","title":"Evaluating ChatGPT-4 Vision on Brazil's National Undergraduate Computer Science Exam","date":"2024-06-14","arxiv_id":"2406.09671","repositories_listed":1,"syntology":null},{"url":"/paper/defan-definitive-answer-dataset-for-llms","slug":"defan-definitive-answer-dataset-for-llms","title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","date":"2024-06-13","arxiv_id":"2406.09155","repositories_listed":1,"syntology":null},{"url":"/paper/ins-mmbench-a-comprehensive-benchmark-for","slug":"ins-mmbench-a-comprehensive-benchmark-for","title":"INS-MMBench: A Comprehensive Benchmark for Evaluating LVLMs' Performance in Insurance","date":"2024-06-13","arxiv_id":"2406.09105","repositories_listed":1,"syntology":null},{"url":"/paper/muirbench-a-comprehensive-benchmark-for","slug":"muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","arxiv_id":"2406.09411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/muirbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.09411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09411"}},"official":{"repos":["muirbench/MuirBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bertaqa-how-much-do-language-models-know","slug":"bertaqa-how-much-do-language-models-know","title":"BertaQA: How Much Do Language Models Know About Local Culture?","date":"2024-06-11","arxiv_id":"2406.07302","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/bertaqa-how-much-do-language-models-know#ran","syntology_url":"https://syntology.ai/paper/2406.07302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07302"}},"official":{"repos":["juletx/bertaqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/open-llm-leaderboard-from-multi-choice-to","slug":"open-llm-leaderboard-from-multi-choice-to","title":"Open-LLM-Leaderboard: From Multi-choice to Open-style Questions for LLMs Evaluation, Benchmark, and Arena","date":"2024-06-11","arxiv_id":"2406.07545","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-llm-leaderboard-from-multi-choice-to#ran","syntology_url":"https://syntology.ai/paper/2406.07545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07545"}},"official":{"repos":["vila-lab/open-llm-leaderboard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-fine-tuning-dataset-and-benchmark-for-large","slug":"a-fine-tuning-dataset-and-benchmark-for-large","title":"A Fine-tuning Dataset and Benchmark for Large Language Models for Protein Understanding","date":"2024-06-08","arxiv_id":"2406.05540","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-fine-tuning-dataset-and-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.05540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05540"}},"official":{"repos":["tsynbio/proteinlmdataset"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-are-not-intelligent-thinkers-introducing","slug":"llms-are-not-intelligent-thinkers-introducing","title":"LLMs Are Not Intelligent Thinkers: Introducing Mathematical Topic Tree Benchmark for Comprehensive Evaluation of LLMs","date":"2024-06-07","arxiv_id":"2406.05194","repositories_listed":1,"syntology":null},{"url":"/paper/every-answer-matters-evaluating-commonsense","slug":"every-answer-matters-evaluating-commonsense","title":"Every Answer Matters: Evaluating Commonsense with Probabilistic Measures","date":"2024-06-06","arxiv_id":"2406.04145","repositories_listed":1,"syntology":null},{"url":"/paper/m-qalm-a-benchmark-to-assess-clinical-reading","slug":"m-qalm-a-benchmark-to-assess-clinical-reading","title":"M-QALM: A Benchmark to Assess Clinical Reading Comprehension and Knowledge Recall in Large Language Models via Question Answering","date":"2024-06-06","arxiv_id":"2406.03699","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-choice-questions-and-large-languages","slug":"multiple-choice-questions-and-large-languages","title":"Multiple Choice Questions and Large Languages Models: A Case Study with Fictional Medical Data","date":"2024-06-04","arxiv_id":"2406.02394","repositories_listed":1,"syntology":null},{"url":"/paper/set-based-prompting-provably-solving-the","slug":"set-based-prompting-provably-solving-the","title":"Order-Independence Without Fine Tuning","date":"2024-06-04","arxiv_id":"2406.06581","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/set-based-prompting-provably-solving-the#ran","syntology_url":"https://syntology.ai/paper/2406.06581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06581"}},"official":{"repos":["reidmcy/set-based-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/topviewrs-vision-language-models-as-top-view","slug":"topviewrs-vision-language-models-as-top-view","title":"TopViewRS: Vision-Language Models as Top-View Spatial Reasoners","date":"2024-06-04","arxiv_id":"2406.02537","repositories_listed":1,"syntology":null},{"url":"/paper/strengthened-symbol-binding-makes-large","slug":"strengthened-symbol-binding-makes-large","title":"Strengthened Symbol Binding Makes Large Language Models Reliable Multiple-Choice Selectors","date":"2024-06-03","arxiv_id":"2406.01026","repositories_listed":1,"syntology":null},{"url":"/paper/an-automatic-question-usability-evaluation","slug":"an-automatic-question-usability-evaluation","title":"An Automatic Question Usability Evaluation Toolkit","date":"2024-05-30","arxiv_id":"2405.20529","repositories_listed":1,"syntology":null},{"url":"/paper/automated-generation-and-tagging-of-knowledge","slug":"automated-generation-and-tagging-of-knowledge","title":"Automated Generation and Tagging of Knowledge Components from Multiple-Choice Questions","date":"2024-05-30","arxiv_id":"2405.20526","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-large-language-model-biases-in","slug":"evaluating-large-language-model-biases-in","title":"Evaluating Large Language Model Biases in Persona-Steered Generation","date":"2024-05-30","arxiv_id":"2405.20253","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-large-language-model-biases-in#ran","syntology_url":"https://syntology.ai/paper/2405.20253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20253"}},"official":{"repos":["andyjliu/persona-steered-generation-bias"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/student-answer-forecasting-transformer-driven","slug":"student-answer-forecasting-transformer-driven","title":"Student Answer Forecasting: Transformer-Driven Answer Choice Prediction for Language Learning","date":"2024-05-30","arxiv_id":"2405.20079","repositories_listed":1,"syntology":null},{"url":"/paper/irel-at-semeval-2024-task-9-improving","slug":"irel-at-semeval-2024-task-9-improving","title":"iREL at SemEval-2024 Task 9: Improving Conventional Prompting Methods for Brain Teasers","date":"2024-05-25","arxiv_id":"2405.16129","repositories_listed":1,"syntology":null},{"url":"/paper/eliciting-informative-text-evaluations-with","slug":"eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","arxiv_id":"2405.15077","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/eliciting-informative-text-evaluations-with#ran","syntology_url":"https://syntology.ai/paper/2405.15077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15077"}},"official":{"repos":["yx-lu/eliciting-informative-text-evaluations-with-large-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-evaluation-of-retrieval-augmented","slug":"automated-evaluation-of-retrieval-augmented","title":"Automated Evaluation of Retrieval-Augmented Language Models with Task-Specific Exam Generation","date":"2024-05-22","arxiv_id":"2405.13622","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-volatility-for-out-of-distribution","slug":"trajectory-volatility-for-out-of-distribution","title":"Embedding Trajectory for Out-of-Distribution Detection in Mathematical Reasoning","date":"2024-05-22","arxiv_id":"2405.14039","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-volatility-for-out-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2405.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14039"}},"official":{"repos":["alsace08/ood-math-reasoning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scifibench-benchmarking-large-multimodal","slug":"scifibench-benchmarking-large-multimodal","title":"SciFIBench: Benchmarking Large Multimodal Models for Scientific Figure Interpretation","date":"2024-05-14","arxiv_id":"2405.08807","repositories_listed":1,"syntology":null},{"url":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/limited-ability-of-llms-to-simulate-human#ran","syntology_url":"https://syntology.ai/paper/2405.07248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07248"}},"official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/throne-an-object-based-hallucination","slug":"throne-an-object-based-hallucination","title":"THRONE: An Object-based Hallucination Benchmark for the Free-form Generations of Large Vision-Language Models","date":"2024-05-08","arxiv_id":"2405.05256","repositories_listed":1,"syntology":null},{"url":"/paper/anchored-answers-unravelling-positional-bias","slug":"anchored-answers-unravelling-positional-bias","title":"Anchored Answers: Unravelling Positional Bias in GPT-2's Multiple-Choice Questions","date":"2024-05-06","arxiv_id":"2405.03205","repositories_listed":1,"syntology":null},{"url":"/paper/self-reflection-in-llm-agents-effects-on","slug":"self-reflection-in-llm-agents-effects-on","title":"Self-Reflection in LLM Agents: Effects on Problem-Solving Performance","date":"2024-05-05","arxiv_id":"2405.06682","repositories_listed":1,"syntology":null},{"url":"/paper/do-large-language-models-understand","slug":"do-large-language-models-understand","title":"Do Large Language Models Understand Conversational Implicature -- A case study with a chinese sitcom","date":"2024-04-30","arxiv_id":"2404.19509","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-large-language-models-understand#ran","syntology_url":"https://syntology.ai/paper/2404.19509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19509"}},"official":{"repos":["sjtu-compling/llm-pragmatics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/can-a-multichoice-dataset-be-repurposed-for","slug":"can-a-multichoice-dataset-be-repurposed-for","title":"From Multiple-Choice to Extractive QA: A Case Study for English and Arabic","date":"2024-04-26","arxiv_id":"2404.17342","repositories_listed":1,"syntology":null},{"url":"/paper/player-enhancing-llm-based-multi-agent","slug":"player-enhancing-llm-based-multi-agent","title":"PLAYER*: Enhancing LLM-based Multi-Agent Communication and Interaction in Murder Mystery Games","date":"2024-04-26","arxiv_id":"2404.17662","repositories_listed":1,"syntology":null},{"url":"/paper/how-far-are-we-to-gpt-4v-closing-the-gap-to","slug":"how-far-are-we-to-gpt-4v-closing-the-gap-to","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","date":"2024-04-25","arxiv_id":"2404.16821","repositories_listed":1,"syntology":null},{"url":"/paper/taxi-evaluating-categorical-knowledge-editing","slug":"taxi-evaluating-categorical-knowledge-editing","title":"TAXI: Evaluating Categorical Knowledge Editing for Language Models","date":"2024-04-23","arxiv_id":"2404.15004","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/taxi-evaluating-categorical-knowledge-editing#ran","syntology_url":"https://syntology.ai/paper/2404.15004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15004"}},"official":{"repos":["derekpowell/taxi"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/unibucllm-harnessing-llms-for-automated","slug":"unibucllm-harnessing-llms-for-automated","title":"UnibucLLM: Harnessing LLMs for Automated Prediction of Item Difficulty and Response Time for Multiple-Choice Questions","date":"2024-04-20","arxiv_id":"2404.13343","repositories_listed":1,"syntology":null},{"url":"/paper/look-at-the-text-instruction-tuned-language","slug":"look-at-the-text-instruction-tuned-language","title":"Look at the Text: Instruction-Tuned Language Models are More Robust Multiple Choice Selectors than You Think","date":"2024-04-12","arxiv_id":"2404.08382","repositories_listed":1,"syntology":null},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mlake-multilingual-knowledge-editing","slug":"mlake-multilingual-knowledge-editing","title":"MLaKE: Multilingual Knowledge Editing Benchmark for Large Language Models","date":"2024-04-07","arxiv_id":"2404.04990","repositories_listed":1,"syntology":null},{"url":"/paper/nlp-at-uc-santa-cruz-at-semeval-2024-task-5","slug":"nlp-at-uc-santa-cruz-at-semeval-2024-task-5","title":"NLP at UC Santa Cruz at SemEval-2024 Task 5: Legal Answer Validation using Few-Shot Multi-Choice QA","date":"2024-04-04","arxiv_id":"2404.03150","repositories_listed":1,"syntology":null},{"url":"/paper/cseprompts-a-benchmark-of-introductory","slug":"cseprompts-a-benchmark-of-introductory","title":"CSEPrompts: A Benchmark of Introductory Computer Science Prompts","date":"2024-04-03","arxiv_id":"2404.02540","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-automated-distractor-generation-for","slug":"exploring-automated-distractor-generation-for","title":"Exploring Automated Distractor Generation for Math Multiple-choice Questions via Large Language Models","date":"2024-04-02","arxiv_id":"2404.02124","repositories_listed":1,"syntology":null},{"url":"/paper/ails-ntua-at-semeval-2024-task-9-cracking","slug":"ails-ntua-at-semeval-2024-task-9-cracking","title":"AILS-NTUA at SemEval-2024 Task 9: Cracking Brain Teasers: Transformer Models for Lateral Thinking Puzzles","date":"2024-04-01","arxiv_id":"2404.01084","repositories_listed":1,"syntology":null},{"url":"/paper/latxa-an-open-language-model-and-evaluation","slug":"latxa-an-open-language-model-and-evaluation","title":"Latxa: An Open Language Model and Evaluation Suite for Basque","date":"2024-03-29","arxiv_id":"2403.20266","repositories_listed":1,"syntology":null},{"url":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","slug":"an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","arxiv_id":"2403.18406","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-image-grid-can-be-worth-a-video-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.18406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18406"}},"official":{"repos":["imagegridworth/IG-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedlm-a-2-7b-parameter-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.18421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18421"}},"official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nl-iti-optimizing-probing-and-intervention","slug":"nl-iti-optimizing-probing-and-intervention","title":"Non-Linear Inference Time Intervention: Improving LLM Truthfulness","date":"2024-03-27","arxiv_id":"2403.18680","repositories_listed":1,"syntology":null},{"url":"/paper/can-multiple-choice-questions-really-be","slug":"can-multiple-choice-questions-really-be","title":"Can multiple-choice questions really be useful in detecting the abilities of LLMs?","date":"2024-03-26","arxiv_id":"2403.17752","repositories_listed":1,"syntology":null},{"url":"/paper/pctoolkit-a-unified-plug-and-play-prompt","slug":"pctoolkit-a-unified-plug-and-play-prompt","title":"PCToolkit: A Unified Plug-and-Play Prompt Compression Toolkit of Large Language Models","date":"2024-03-26","arxiv_id":"2403.17411","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-long-videos-in-one-multimodal","slug":"understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","arxiv_id":"2403.16998","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-long-videos-in-one-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.16998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16998"}},"official":{"repos":["kahnchana/mvu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pragmatic-competence-evaluation-of-large","slug":"pragmatic-competence-evaluation-of-large","title":"Pragmatic Competence Evaluation of Large Language Models for the Korean Language","date":"2024-03-19","arxiv_id":"2403.12675","repositories_listed":1,"syntology":null},{"url":"/paper/exams-v-a-multi-discipline-multilingual","slug":"exams-v-a-multi-discipline-multilingual","title":"EXAMS-V: A Multi-Discipline Multilingual Multimodal Exam Benchmark for Evaluating Vision Language Models","date":"2024-03-15","arxiv_id":"2403.10378","repositories_listed":1,"syntology":null},{"url":"/paper/towards-diverse-perspective-learning-with","slug":"towards-diverse-perspective-learning-with","title":"Towards Diverse Perspective Learning with Selection over Multiple Temporal Poolings","date":"2024-03-14","arxiv_id":"2403.09749","repositories_listed":1,"syntology":null},{"url":"/paper/complex-reasoning-over-logical-queries-on","slug":"complex-reasoning-over-logical-queries-on","title":"Complex Reasoning over Logical Queries on Commonsense Knowledge Graphs","date":"2024-03-12","arxiv_id":"2403.07398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/complex-reasoning-over-logical-queries-on#ran","syntology_url":"https://syntology.ai/paper/2403.07398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07398"}},"official":{"repos":["tqfang/complex-commonsense-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unfamiliar-finetuning-examples-control-how","slug":"unfamiliar-finetuning-examples-control-how","title":"Unfamiliar Finetuning Examples Control How Language Models Hallucinate","date":"2024-03-08","arxiv_id":"2403.05612","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unfamiliar-finetuning-examples-control-how#ran","syntology_url":"https://syntology.ai/paper/2403.05612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05612"}},"official":{"repos":["katiekang1998/llm_hallucinations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/to-generate-or-to-retrieve-on-the","slug":"to-generate-or-to-retrieve-on-the","title":"To Generate or to Retrieve? On the Effectiveness of Artificial Contexts for Medical Open-Domain Question Answering","date":"2024-03-04","arxiv_id":"2403.01924","repositories_listed":1,"syntology":null},{"url":"/paper/parallelparc-a-scalable-pipeline-for","slug":"parallelparc-a-scalable-pipeline-for","title":"ParallelPARC: A Scalable Pipeline for Generating Natural-Language Analogies","date":"2024-03-02","arxiv_id":"2403.01139","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallelparc-a-scalable-pipeline-for#ran","syntology_url":"https://syntology.ai/paper/2403.01139","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01139"}},"official":{"repos":["orensul/parallelparc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/newsbench-systematic-evaluation-of-llms-for","slug":"newsbench-systematic-evaluation-of-llms-for","title":"NewsBench: A Systematic Evaluation Framework for Assessing Editorial Capabilities of Large Language Models in Chinese Journalism","date":"2024-02-29","arxiv_id":"2403.00862","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/newsbench-systematic-evaluation-of-llms-for#ran","syntology_url":"https://syntology.ai/paper/2403.00862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00862"}},"official":{"repos":["iaar-shanghai/newsbench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/nextlevelbert-investigating-masked-language","slug":"nextlevelbert-investigating-masked-language","title":"NextLevelBERT: Masked Language Modeling with Higher-Level Representations for Long Documents","date":"2024-02-27","arxiv_id":"2402.17682","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-models-for-learning","slug":"leveraging-large-language-models-for-learning","title":"Leveraging Large Language Models for Learning Complex Legal Concepts through Storytelling","date":"2024-02-26","arxiv_id":"2402.17019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-large-language-models-for-learning#ran","syntology_url":"https://syntology.ai/paper/2402.17019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17019"}},"official":{"repos":["hjian42/legalstories"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mozip-a-multilingual-benchmark-to-evaluate","slug":"mozip-a-multilingual-benchmark-to-evaluate","title":"MoZIP: A Multilingual Benchmark to Evaluate Large Language Models in Intellectual Property","date":"2024-02-26","arxiv_id":"2402.16389","repositories_listed":1,"syntology":null},{"url":"/paper/political-compass-or-spinning-arrow-towards","slug":"political-compass-or-spinning-arrow-towards","title":"Political Compass or Spinning Arrow? Towards More Meaningful Evaluations for Values and Opinions in Large Language Models","date":"2024-02-26","arxiv_id":"2402.16786","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/political-compass-or-spinning-arrow-towards#ran","syntology_url":"https://syntology.ai/paper/2402.16786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16786"}},"official":{"repos":["paul-rottger/llm-values-pct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sportqa-a-benchmark-for-sports-understanding","slug":"sportqa-a-benchmark-for-sports-understanding","title":"SportQA: A Benchmark for Sports Understanding in Large Language Models","date":"2024-02-24","arxiv_id":"2402.15862","repositories_listed":1,"syntology":null},{"url":"/paper/biomedical-entity-linking-as-multiple-choice","slug":"biomedical-entity-linking-as-multiple-choice","title":"Biomedical Entity Linking as Multiple Choice Question Answering","date":"2024-02-23","arxiv_id":"2402.15189","repositories_listed":1,"syntology":null},{"url":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tombench-benchmarking-theory-of-mind-in-large#ran","syntology_url":"https://syntology.ai/paper/2402.15052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15052"}},"official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/my-answer-is-c-first-token-probabilities-do","slug":"my-answer-is-c-first-token-probabilities-do","title":"\"My Answer is C\": First-Token Probabilities Do Not Match Text Answers in Instruction-Tuned Language Models","date":"2024-02-22","arxiv_id":"2402.14499","repositories_listed":1,"syntology":null}],"record_sha256":"52b4a41d8c1bc824c3ce730532e08d8a58dc977687852ff49d45085e70309a46","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}