{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/4","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":12,"rows_per_page":100,"rows":[301,400],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/3","next":"/task/multiple-choice/papers/5","papers":[{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/arabicmmlu-assessing-massive-multitask","slug":"arabicmmlu-assessing-massive-multitask","title":"ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic","date":"2024-02-20","arxiv_id":"2402.12840","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/arabicmmlu-assessing-massive-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.12840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12840"}},"official":{"repos":["mbzuai-nlp/arabicmmlu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/bimedix-bilingual-medical-mixture-of-experts","slug":"bimedix-bilingual-medical-mixture-of-experts","title":"BiMediX: Bilingual Medical Mixture of Experts LLM","date":"2024-02-20","arxiv_id":"2402.13253","repositories_listed":1,"syntology":null},{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","slug":"artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/artifacts-or-abduction-how-do-llms-answer#ran","syntology_url":"https://syntology.ai/paper/2402.12483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12483"}},"official":{"repos":["nbalepur/mcqa-artifacts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-quantification-in-fine-tuned-llms","slug":"uncertainty-quantification-in-fine-tuned-llms","title":"Uncertainty quantification in fine-tuned LLMs using LoRA ensembles","date":"2024-02-19","arxiv_id":"2402.12264","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncertainty-quantification-in-fine-tuned-llms#ran","syntology_url":"https://syntology.ai/paper/2402.12264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12264"}},"official":{"repos":["oleksandr-balabanov/equivariant-posteriors"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/question-instructed-visual-descriptions-for","slug":"question-instructed-visual-descriptions-for","title":"Question-Instructed Visual Descriptions for Zero-Shot Video Question Answering","date":"2024-02-16","arxiv_id":"2402.10698","repositories_listed":1,"syntology":null},{"url":"/paper/cybermetric-a-benchmark-dataset-for","slug":"cybermetric-a-benchmark-dataset-for","title":"CyberMetric: A Benchmark Dataset based on Retrieval-Augmented Generation for Evaluating LLMs in Cybersecurity Knowledge","date":"2024-02-12","arxiv_id":"2402.07688","repositories_listed":1,"syntology":null},{"url":"/paper/salad-bench-a-hierarchical-and-comprehensive","slug":"salad-bench-a-hierarchical-and-comprehensive","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","date":"2024-02-07","arxiv_id":"2402.05044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/salad-bench-a-hierarchical-and-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.05044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05044"}},"official":{"repos":["opensafetylab/salad-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-effect-of-sampling-temperature-on-problem","slug":"the-effect-of-sampling-temperature-on-problem","title":"The Effect of Sampling Temperature on Problem Solving in Large Language Models","date":"2024-02-07","arxiv_id":"2402.05201","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-effect-of-sampling-temperature-on-problem#ran","syntology_url":"https://syntology.ai/paper/2402.05201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05201"}},"official":{"repos":["matthewrenze/jhu-llm-temperature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-textbook-question-answering-task","slug":"enhancing-textbook-question-answering-task","title":"Enhancing textual textbook question answering with large language models and retrieval augmented generation","date":"2024-02-05","arxiv_id":"2402.05128","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/enhancing-textbook-question-answering-task#ran","syntology_url":"https://syntology.ai/paper/2402.05128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05128"}},"official":{"repos":["hessaalawwad/plr-tqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/an-information-theoretic-approach-to-analyze","slug":"an-information-theoretic-approach-to-analyze","title":"An Information-Theoretic Approach to Analyze NLP Classification Tasks","date":"2024-02-01","arxiv_id":"2402.00978","repositories_listed":1,"syntology":null},{"url":"/paper/when-benchmarks-are-targets-revealing-the","slug":"when-benchmarks-are-targets-revealing-the","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","date":"2024-02-01","arxiv_id":"2402.01781","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-benchmarks-are-targets-revealing-the#ran","syntology_url":"https://syntology.ai/paper/2402.01781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01781"}},"official":{"repos":["national-center-for-ai-saudi-arabia/lm-evaluation-harness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/e-eval-a-comprehensive-chinese-k-12-education","slug":"e-eval-a-comprehensive-chinese-k-12-education","title":"E-EVAL: A Comprehensive Chinese K-12 Education Evaluation Benchmark for Large Language Models","date":"2024-01-29","arxiv_id":"2401.15927","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/e-eval-a-comprehensive-chinese-k-12-education#ran","syntology_url":"https://syntology.ai/paper/2401.15927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15927"}},"official":{"repos":["ai-edu-lab/e-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-medical-reasoning-through-retrieval","slug":"improving-medical-reasoning-through-retrieval","title":"Improving Medical Reasoning through Retrieval and Self-Reflection with Retrieval-Augmented Large Language Models","date":"2024-01-27","arxiv_id":"2401.15269","repositories_listed":1,"syntology":null},{"url":"/paper/cmmu-a-benchmark-for-chinese-multi-modal","slug":"cmmu-a-benchmark-for-chinese-multi-modal","title":"CMMU: A Benchmark for Chinese Multi-modal Multi-type Question Understanding and Reasoning","date":"2024-01-25","arxiv_id":"2401.14011","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmmu-a-benchmark-for-chinese-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.14011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14011"}},"official":{"repos":["flagopen/cmmu"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longhealth-a-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.14490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14490"}},"official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-study-on-large-language-models-limitations","slug":"a-study-on-large-language-models-limitations","title":"A Study on Large Language Models' Limitations in Multiple-Choice Question Answering","date":"2024-01-15","arxiv_id":"2401.07955","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-external-knowledge-resources-to","slug":"leveraging-external-knowledge-resources-to","title":"Towards Efficient Methods in Medical Question Answering using Knowledge Graph Embeddings","date":"2024-01-15","arxiv_id":"2401.07977","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-external-knowledge-resources-to#ran","syntology_url":"https://syntology.ai/paper/2401.07977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07977"}},"official":{"repos":["saptarshi059/cdqa-project"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-novel-multi-stage-prompting-approach-for","slug":"a-novel-multi-stage-prompting-approach-for","title":"A Novel Multi-Stage Prompting Approach for Language Agnostic MCQ Generation using GPT","date":"2024-01-13","arxiv_id":"2401.07098","repositories_listed":1,"syntology":null},{"url":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-benefits-of-a-concise-chain-of-thought-on#ran","syntology_url":"https://syntology.ai/paper/2401.05618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05618"}},"official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-bench-benchmarking-multimodal-large","slug":"seed-bench-benchmarking-multimodal-large","title":"SEED-Bench: Benchmarking Multimodal Large Language Models","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/roleeval-a-bilingual-role-evaluation","slug":"roleeval-a-bilingual-role-evaluation","title":"RoleEval: A Bilingual Role Evaluation Benchmark for Large Language Models","date":"2023-12-26","arxiv_id":"2312.16132","repositories_listed":1,"syntology":null},{"url":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/secqa-a-concise-question-answering-dataset#ran","syntology_url":"https://syntology.ai/paper/2312.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15838"}},"official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/think-and-retrieval-a-hypothesis-knowledge","slug":"think-and-retrieval-a-hypothesis-knowledge","title":"HyKGE: A Hypothesis Knowledge Graph Enhanced Framework for Accurate and Reliable Medical LLMs Responses","date":"2023-12-26","arxiv_id":"2312.15883","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-unified-multimodal-reasoning","slug":"towards-a-unified-multimodal-reasoning","title":"Towards a Unified Multimodal Reasoning Framework","date":"2023-12-22","arxiv_id":"2312.15021","repositories_listed":1,"syntology":null},{"url":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-in-depth-look-at-gemini-s-language#ran","syntology_url":"https://syntology.ai/paper/2312.11444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11444"}},"official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-hypothesis-dropout-estimating-the","slug":"multiple-hypothesis-dropout-estimating-the","title":"Multiple Hypothesis Dropout: Estimating the Parameters of Multi-Modal Output Distributions","date":"2023-12-18","arxiv_id":"2312.11735","repositories_listed":1,"syntology":null},{"url":"/paper/marathon-a-race-through-the-realm-of-long","slug":"marathon-a-race-through-the-realm-of-long","title":"Marathon: A Race Through the Realm of Long Context with Large Language Models","date":"2023-12-15","arxiv_id":"2312.09542","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-surface-probing-llama-across-scales","slug":"beyond-surface-probing-llama-across-scales","title":"Is Bigger and Deeper Always Better? Probing LLaMA Across Scales and Layers","date":"2023-12-07","arxiv_id":"2312.04333","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-surface-probing-llama-across-scales#ran","syntology_url":"https://syntology.ai/paper/2312.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04333"}},"official":{"repos":["nuochenpku/llama_analysis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explanatory-argument-extraction-of-correct","slug":"explanatory-argument-extraction-of-correct","title":"Explanatory Argument Extraction of Correct Answers in Resident Medical Exams","date":"2023-12-01","arxiv_id":"2312.00567","repositories_listed":1,"syntology":null},{"url":"/paper/biomedical-knowledge-graph-enhanced-prompt","slug":"biomedical-knowledge-graph-enhanced-prompt","title":"Biomedical knowledge graph-optimized prompt generation for large language models","date":"2023-11-29","arxiv_id":"2311.17330","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/biomedical-knowledge-graph-enhanced-prompt#ran","syntology_url":"https://syntology.ai/paper/2311.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17330"}},"official":{"repos":["BaranziniLab/KG_RAG"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/clomo-counterfactual-logical-modification","slug":"clomo-counterfactual-logical-modification","title":"CLOMO: Counterfactual Logical Modification with Large Language Models","date":"2023-11-29","arxiv_id":"2311.17438","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clomo-counterfactual-logical-modification#ran","syntology_url":"https://syntology.ai/paper/2311.17438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17438"}},"official":{"repos":["eleanor-h/clomo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/performance-trade-offs-of-watermarking-large","slug":"performance-trade-offs-of-watermarking-large","title":"Downstream Trade-offs of a Family of Text Watermarks","date":"2023-11-16","arxiv_id":"2311.09816","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/performance-trade-offs-of-watermarking-large#ran","syntology_url":"https://syntology.ai/paper/2311.09816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09816"}},"official":{"repos":["flair-iisc/watermark_tradeoffs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/it-s-not-easy-being-wrong-evaluating-process","slug":"it-s-not-easy-being-wrong-evaluating-process","title":"It's Not Easy Being Wrong: Large Language Models Struggle with Process of Elimination Reasoning","date":"2023-11-13","arxiv_id":"2311.07532","repositories_listed":1,"syntology":null},{"url":"/paper/fake-alignment-are-llms-really-aligned-well","slug":"fake-alignment-are-llms-really-aligned-well","title":"Fake Alignment: Are LLMs Really Aligned Well?","date":"2023-11-10","arxiv_id":"2311.05915","repositories_listed":1,"syntology":null},{"url":"/paper/resilient-multiple-choice-learning-a-learned","slug":"resilient-multiple-choice-learning-a-learned","title":"Resilient Multiple Choice Learning: A learned scoring scheme with application to audio scene analysis","date":"2023-11-02","arxiv_id":"2311.01052","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/resilient-multiple-choice-learning-a-learned#ran","syntology_url":"https://syntology.ai/paper/2311.01052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01052"}},"official":{"repos":["victorletzelter/code-rmcl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-open-source-data-contamination-report-for","slug":"an-open-source-data-contamination-report-for","title":"An Open Source Data Contamination Report for Large Language Models","date":"2023-10-26","arxiv_id":"2310.17589","repositories_listed":1,"syntology":null},{"url":"/paper/poe-process-of-elimination-for-multiple","slug":"poe-process-of-elimination-for-multiple","title":"POE: Process of Elimination for Multiple Choice Reasoning","date":"2023-10-24","arxiv_id":"2310.15575","repositories_listed":1,"syntology":null},{"url":"/paper/storyanalogy-deriving-story-level-analogies","slug":"storyanalogy-deriving-story-level-analogies","title":"StoryAnalogy: Deriving Story-level Analogies from Large Language Models to Unlock Analogical Understanding","date":"2023-10-19","arxiv_id":"2310.12874","repositories_listed":1,"syntology":null},{"url":"/paper/jmedlora-medical-domain-adaptation-on","slug":"jmedlora-medical-domain-adaptation-on","title":"JMedLoRA:Medical Domain Adaptation on Japanese Large Language Models using Instruction-tuning","date":"2023-10-16","arxiv_id":"2310.10083","repositories_listed":1,"syntology":null},{"url":"/paper/kgquiz-evaluating-the-generalization-of","slug":"kgquiz-evaluating-the-generalization-of","title":"KGQuiz: Evaluating the Generalization of Encoded Knowledge in Large Language Models","date":"2023-10-15","arxiv_id":"2310.09725","repositories_listed":1,"syntology":null},{"url":"/paper/opseval-a-comprehensive-task-oriented-aiops","slug":"opseval-a-comprehensive-task-oriented-aiops","title":"OpsEval: A Comprehensive IT Operations Benchmark Suite for Large Language Models","date":"2023-10-11","arxiv_id":"2310.07637","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-zero-shot-abilities-of-vision","slug":"analyzing-zero-shot-abilities-of-vision","title":"Analyzing Zero-Shot Abilities of Vision-Language Models on Video Understanding Tasks","date":"2023-10-07","arxiv_id":"2310.04914","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-multi-agent-coordination-abilities","slug":"evaluating-multi-agent-coordination-abilities","title":"LLM-Coordination: Evaluating and Analyzing Multi-agent Coordination Abilities in Large Language Models","date":"2023-10-05","arxiv_id":"2310.03903","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-multi-agent-coordination-abilities#ran","syntology_url":"https://syntology.ai/paper/2310.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03903"}},"official":{"repos":["eric-ai-lab/llm_coordination"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/autocast-enhancing-world-event-prediction","slug":"autocast-enhancing-world-event-prediction","title":"AutoCast++: Enhancing World Event Prediction with Zero-shot Ranking-based Context Retrieval","date":"2023-10-03","arxiv_id":"2310.01880","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/autocast-enhancing-world-event-prediction#ran","syntology_url":"https://syntology.ai/paper/2310.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01880"}},"official":{"repos":["BorealisAI/Autocast-plus-plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-provide-security","slug":"can-large-language-models-provide-security","title":"Can Large Language Models Provide Security & Privacy Advice? Measuring the Ability of LLMs to Refute Misconceptions","date":"2023-10-03","arxiv_id":"2310.02431","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-knowledge-bases-for-visual","slug":"language-models-as-knowledge-bases-for-visual","title":"Language Models as Knowledge Bases for Visual Word Sense Disambiguation","date":"2023-10-03","arxiv_id":"2310.01960","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-models-as-knowledge-bases-for-visual#ran","syntology_url":"https://syntology.ai/paper/2310.01960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01960"}},"official":{"repos":["anastasiakrith/llm-for-vwsd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fool-your-vision-and-language-model-with","slug":"fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","arxiv_id":"2310.01651","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fool-your-vision-and-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2310.01651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01651"}},"official":{"repos":["ys-zong/foolyourvllms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fusing-models-with-complementary-expertise","slug":"fusing-models-with-complementary-expertise","title":"Fusing Models with Complementary Expertise","date":"2023-10-02","arxiv_id":"2310.01542","repositories_listed":1,"syntology":{"n":19,"n_ran":13,"n_constructed":2,"n_ran_checked":6,"n_instrument":7,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":19,"phrase":"13 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 7 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fusing-models-with-complementary-expertise#ran","syntology_url":"https://syntology.ai/paper/2310.01542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01542"}},"official":{"repos":["hwang595/foe-iclr2024"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/estimating-contamination-via-perplexity","slug":"estimating-contamination-via-perplexity","title":"Estimating Contamination via Perplexity: Quantifying Memorisation in Language Model Evaluation","date":"2023-09-19","arxiv_id":"2309.10677","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-self-reinforcement-for-improving","slug":"exploring-self-reinforcement-for-improving","title":"Exploring Iterative Enhancement for Improving Learnersourced Multiple-Choice Question Explanations with Large Language Models","date":"2023-09-19","arxiv_id":"2309.10444","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-self-reinforcement-for-improving#ran","syntology_url":"https://syntology.ai/paper/2309.10444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10444"}},"official":{"repos":["strong-ai-lab/explanation-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safetybench-evaluating-the-safety-of-large","slug":"safetybench-evaluating-the-safety-of-large","title":"SafetyBench: Evaluating the Safety of Large Language Models","date":"2023-09-13","arxiv_id":"2309.07045","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safetybench-evaluating-the-safety-of-large#ran","syntology_url":"https://syntology.ai/paper/2309.07045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07045"}},"official":{"repos":["thu-coai/safetybench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-large-language-models-selection-bias-in","slug":"on-large-language-models-selection-bias-in","title":"Large Language Models Are Not Robust Multiple Choice Selectors","date":"2023-09-07","arxiv_id":"2309.03882","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-large-language-models-selection-bias-in#ran","syntology_url":"https://syntology.ai/paper/2309.03882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03882"}},"official":{"repos":["chujiezheng/llm-mcq-bias"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codeapex-a-bilingual-programming-evaluation","slug":"codeapex-a-bilingual-programming-evaluation","title":"CodeApex: A Bilingual Programming Evaluation Benchmark for Large Language Models","date":"2023-09-05","arxiv_id":"2309.01940","repositories_listed":1,"syntology":null},{"url":"/paper/inceptnet-precise-and-early-disease-detection","slug":"inceptnet-precise-and-early-disease-detection","title":"INCEPTNET: Precise And Early Disease Detection Application For Medical Images Analyses","date":"2023-09-05","arxiv_id":"2309.02147","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-intelligence-of-large","slug":"spoken-language-intelligence-of-large","title":"Spoken Language Intelligence of Large Language Models for Language Learning","date":"2023-08-28","arxiv_id":"2308.14536","repositories_listed":1,"syntology":null},{"url":"/paper/librisqa-pioneering-free-form-and-open-ended","slug":"librisqa-pioneering-free-form-and-open-ended","title":"LibriSQA: A Novel Dataset and Framework for Spoken Question Answering with Large Language Models","date":"2023-08-20","arxiv_id":"2308.10390","repositories_listed":1,"syntology":null},{"url":"/paper/fineval-a-chinese-financial-domain-knowledge","slug":"fineval-a-chinese-financial-domain-knowledge","title":"FinEval: A Chinese Financial Domain Knowledge Evaluation Benchmark for Large Language Models","date":"2023-08-19","arxiv_id":"2308.09975","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fineval-a-chinese-financial-domain-knowledge#ran","syntology_url":"https://syntology.ai/paper/2308.09975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09975"}},"official":{"repos":["sufe-aiflm-lab/fineval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-automated-distractor-and-feedback","slug":"exploring-automated-distractor-and-feedback","title":"Automated Distractor and Feedback Generation for Math Multiple-choice Questions via In-context Learning","date":"2023-08-07","arxiv_id":"2308.03234","repositories_listed":1,"syntology":null},{"url":"/paper/chatgpt-for-gtfs-from-words-to-information","slug":"chatgpt-for-gtfs-from-words-to-information","title":"ChatGPT for GTFS: Benchmarking LLMs on GTFS Understanding and Retrieval","date":"2023-08-04","arxiv_id":"2308.02618","repositories_listed":1,"syntology":null},{"url":"/paper/recomif-reading-comprehension-based-multi","slug":"recomif-reading-comprehension-based-multi","title":"ReCoMIF: Reading comprehension based multi-source information fusion network for Chinese spoken language understanding","date":"2023-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/moviechat-from-dense-token-to-sparse-memory","slug":"moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","arxiv_id":"2307.16449","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-language-model-assisted-education","slug":"a-large-language-model-assisted-education","title":"A large language model-assisted education tool to provide feedback on open-ended responses","date":"2023-07-25","arxiv_id":"2308.02439","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-human-like-multi-modal-reasoning-a","slug":"enhancing-human-like-multi-modal-reasoning-a","title":"Enhancing Human-like Multi-Modal Reasoning: A New Challenging Dataset and Comprehensive Framework","date":"2023-07-24","arxiv_id":"2307.12626","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-human-like-multi-modal-reasoning-a#ran","syntology_url":"https://syntology.ai/paper/2307.12626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12626"}},"official":{"repos":["weijingxuan/COCO-MMR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scibench-evaluating-college-level-scientific","slug":"scibench-evaluating-college-level-scientific","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","date":"2023-07-20","arxiv_id":"2307.10635","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scibench-evaluating-college-level-scientific#ran","syntology_url":"https://syntology.ai/paper/2307.10635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10635"}},"official":{"repos":["mandyyyyii/scibench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-the-quality-of-multiple-choice","slug":"assessing-the-quality-of-multiple-choice","title":"Assessing the Quality of Multiple-Choice Questions Using GPT-4 and Rule-Based Methods","date":"2023-07-16","arxiv_id":"2307.08161","repositories_listed":1,"syntology":null},{"url":"/paper/structured-dialogue-discourse-parsing-1","slug":"structured-dialogue-discourse-parsing-1","title":"Structured Dialogue Discourse Parsing","date":"2023-06-26","arxiv_id":"2306.15103","repositories_listed":1,"syntology":null},{"url":"/paper/solving-and-generating-npr-sunday-puzzles","slug":"solving-and-generating-npr-sunday-puzzles","title":"Solving and Generating NPR Sunday Puzzles with Large Language Models","date":"2023-06-21","arxiv_id":"2306.12255","repositories_listed":1,"syntology":null},{"url":"/paper/questioning-the-survey-responses-of-large","slug":"questioning-the-survey-responses-of-large","title":"Questioning the Survey Responses of Large Language Models","date":"2023-06-13","arxiv_id":"2306.07951","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/questioning-the-survey-responses-of-large#ran","syntology_url":"https://syntology.ai/paper/2306.07951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07951"}},"official":{"repos":["socialfoundations/surveying-language-models"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-prediction-with-large-language","slug":"conformal-prediction-with-large-language","title":"Conformal Prediction with Large Language Models for Multi-Choice Question Answering","date":"2023-05-28","arxiv_id":"2305.18404","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/conformal-prediction-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.18404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18404"}},"official":{"repos":["bhaweshiitk/conformalllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/attentiveness-to-answer-choices-doesn-t","slug":"attentiveness-to-answer-choices-doesn-t","title":"Increasing Probability Mass on Answer Choices Does Not Always Improve Accuracy","date":"2023-05-24","arxiv_id":"2305.14596","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/attentiveness-to-answer-choices-doesn-t#ran","syntology_url":"https://syntology.ai/paper/2305.14596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14596"}},"official":{"repos":["allenai/revisiting_surface_form_competition"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/this-land-is-your-my-land-evaluating","slug":"this-land-is-your-my-land-evaluating","title":"This Land is {Your, My} Land: Evaluating Geopolitical Biases in Language Models","date":"2023-05-24","arxiv_id":"2305.14610","repositories_listed":1,"syntology":null},{"url":"/paper/tomchallenges-a-principle-guided-dataset-and","slug":"tomchallenges-a-principle-guided-dataset-and","title":"ToMChallenges: A Principle-Guided Dataset and Diverse Evaluation Tasks for Exploring Theory of Mind","date":"2023-05-24","arxiv_id":"2305.15068","repositories_listed":1,"syntology":null},{"url":"/paper/narrative-xl-a-large-scale-dataset-for-long","slug":"narrative-xl-a-large-scale-dataset-for-long","title":"NarrativeXL: A Large-scale Dataset For Long-Term Memory Models","date":"2023-05-23","arxiv_id":"2305.13877","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-forward-tuning-boosts-in-context","slug":"iterative-forward-tuning-boosts-in-context","title":"Iterative Forward Tuning Boosts In-Context Learning in Language Models","date":"2023-05-22","arxiv_id":"2305.13016","repositories_listed":1,"syntology":null},{"url":"/paper/vnhsge-vietnamese-high-school-graduation","slug":"vnhsge-vietnamese-high-school-graduation","title":"VNHSGE: VietNamese High School Graduation Examination Dataset for Large Language Models","date":"2023-05-20","arxiv_id":"2305.12199","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantitative-study-of-nlp-approaches-to","slug":"a-quantitative-study-of-nlp-approaches-to","title":"A quantitative study of NLP approaches to question difficulty estimation","date":"2023-05-17","arxiv_id":"2305.10236","repositories_listed":1,"syntology":null},{"url":"/paper/m3ke-a-massive-multi-level-multi-subject","slug":"m3ke-a-massive-multi-level-multi-subject","title":"M3KE: A Massive Multi-Level Multi-Subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2023-05-17","arxiv_id":"2305.10263","repositories_listed":1,"syntology":null},{"url":"/paper/c-eval-a-multi-level-multi-discipline-chinese-1","slug":"c-eval-a-multi-level-multi-discipline-chinese-1","title":"C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models","date":"2023-05-15","arxiv_id":"2305.08322","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/c-eval-a-multi-level-multi-discipline-chinese-1#ran","syntology_url":"https://syntology.ai/paper/2305.08322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08322"}},"official":{"repos":["hkust-nlp/ceval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embrace-evaluation-and-modifications-for","slug":"embrace-evaluation-and-modifications-for","title":"EMBRACE: Evaluation and Modifications for Boosting RACE","date":"2023-05-15","arxiv_id":"2305.08433","repositories_listed":1,"syntology":null},{"url":"/paper/frenchmedmcqa-a-french-multiple-choice-1","slug":"frenchmedmcqa-a-french-multiple-choice-1","title":"FrenchMedMCQA: A French Multiple-Choice Question Answering Dataset for Medical domain","date":"2023-04-09","arxiv_id":"2304.04280","repositories_listed":1,"syntology":null},{"url":"/paper/a-multiple-choices-reading-comprehension","slug":"a-multiple-choices-reading-comprehension","title":"A Multiple Choices Reading Comprehension Corpus for Vietnamese Language Education","date":"2023-03-31","arxiv_id":"2303.18162","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-gpt-3-5-and-gpt-4-models-on","slug":"evaluating-gpt-3-5-and-gpt-4-models-on","title":"Evaluating GPT-3.5 and GPT-4 Models on Brazilian University Admission Exams","date":"2023-03-29","arxiv_id":"2303.17003","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-gpt-3-5-and-gpt-4-models-on#ran","syntology_url":"https://syntology.ai/paper/2303.17003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17003"}},"official":{"repos":["piresramon/gpt-4-enem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/long-horizon-temperature-scaling","slug":"long-horizon-temperature-scaling","title":"Long Horizon Temperature Scaling","date":"2023-02-07","arxiv_id":"2302.03686","repositories_listed":1,"syntology":null},{"url":"/paper/padl-language-directed-physics-based","slug":"padl-language-directed-physics-based","title":"PADL: Language-Directed Physics-Based Character Control","date":"2023-01-31","arxiv_id":"2301.13868","repositories_listed":1,"syntology":null},{"url":"/paper/gpt-as-knowledge-worker-a-zero-shot","slug":"gpt-as-knowledge-worker-a-zero-shot","title":"GPT as Knowledge Worker: A Zero-Shot Evaluation of (AI)CPA Capabilities","date":"2023-01-11","arxiv_id":"2301.04408","repositories_listed":1,"syntology":null},{"url":"/paper/mind-reasoning-manners-enhancing-type","slug":"mind-reasoning-manners-enhancing-type","title":"Mind Reasoning Manners: Enhancing Type Perception for Generalized Zero-shot Logical Reasoning over Text","date":"2023-01-08","arxiv_id":"2301.02983","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-encode-clinical","slug":"large-language-models-encode-clinical","title":"Large Language Models Encode Clinical Knowledge","date":"2022-12-26","arxiv_id":"2212.13138","repositories_listed":1,"syntology":null},{"url":"/paper/training-trajectories-of-language-models","slug":"training-trajectories-of-language-models","title":"Training Trajectories of Language Models Across Scales","date":"2022-12-19","arxiv_id":"2212.09803","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-background-knowledge-for-robust","slug":"utilizing-background-knowledge-for-robust","title":"Utilizing Background Knowledge for Robust Reasoning over Traffic Situations","date":"2022-12-04","arxiv_id":"2212.07798","repositories_listed":1,"syntology":null},{"url":"/paper/which-shortcut-solution-do-question-answering","slug":"which-shortcut-solution-do-question-answering","title":"Which Shortcut Solution Do Question Answering Models Prefer to Learn?","date":"2022-11-29","arxiv_id":"2211.16220","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-knowledge-dependency-of","slug":"evaluating-the-knowledge-dependency-of","title":"Evaluating the Knowledge Dependency of Questions","date":"2022-11-21","arxiv_id":"2211.11902","repositories_listed":1,"syntology":null},{"url":"/paper/unified-question-answering-in-slovene","slug":"unified-question-answering-in-slovene","title":"Unified Question Answering in Slovene","date":"2022-11-16","arxiv_id":"2211.09159","repositories_listed":1,"syntology":null},{"url":"/paper/world-knowledge-in-multiple-choice-reading","slug":"world-knowledge-in-multiple-choice-reading","title":"World Knowledge in Multiple Choice Reading Comprehension","date":"2022-11-13","arxiv_id":"2211.07040","repositories_listed":1,"syntology":null},{"url":"/paper/advertising-strategy-for-profit-maximization","slug":"advertising-strategy-for-profit-maximization","title":"A Profit-Maximizing Strategy for Advertising on the e-Commerce Platforms","date":"2022-10-31","arxiv_id":"2211.01160","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reuse-distractors-to-support","slug":"learning-to-reuse-distractors-to-support","title":"Learning to Reuse Distractors to support Multiple Choice Question Generation in Education","date":"2022-10-25","arxiv_id":"2210.13964","repositories_listed":1,"syntology":null},{"url":"/paper/cascading-biases-investigating-the-effect-of","slug":"cascading-biases-investigating-the-effect-of","title":"Cascading Biases: Investigating the Effect of Heuristic Annotation Strategies on Data and Models","date":"2022-10-24","arxiv_id":"2210.13439","repositories_listed":1,"syntology":null}],"record_sha256":"19a6dbd1eeaa9b684cf74fa16a437965cfe12511e70adc20212cea7620f2624f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}