{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/5","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":12,"rows_per_page":100,"rows":[401,500],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/4","next":"/task/multiple-choice/papers/6","papers":[{"url":"/paper/leveraging-large-language-models-for-multiple","slug":"leveraging-large-language-models-for-multiple","title":"Leveraging Large Language Models for Multiple Choice Question Answering","date":"2022-10-22","arxiv_id":"2210.12353","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-large-language-models-for-multiple#ran","syntology_url":"https://syntology.ai/paper/2210.12353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12353"}},"official":{"repos":["byu-pccl/leveraging-llms-for-mcqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/perception-test-a-diagnostic-benchmark-for","slug":"perception-test-a-diagnostic-benchmark-for","title":"Perception Test: A Diagnostic Benchmark for Multimodal Models","date":"2022-10-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-learners-for-natural-language","slug":"zero-shot-learners-for-natural-language","title":"Zero-Shot Learners for Natural Language Understanding via a Unified Multiple Choice Perspective","date":"2022-10-16","arxiv_id":"2210.08590","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-automated-answer-scoring","slug":"real-time-automated-answer-scoring","title":"Real-Time Automated Answer Scoring","date":"2022-10-13","arxiv_id":"2210.09004","repositories_listed":1,"syntology":null},{"url":"/paper/eduqg-a-multi-format-multiple-choice-dataset","slug":"eduqg-a-multi-format-multiple-choice-dataset","title":"EduQG: A Multi-format Multiple Choice Dataset for the Educational Domain","date":"2022-10-12","arxiv_id":"2210.06104","repositories_listed":1,"syntology":null},{"url":"/paper/can-we-guide-a-multi-hop-reasoning-language","slug":"can-we-guide-a-multi-hop-reasoning-language","title":"Can We Guide a Multi-Hop Reasoning Language Model to Incrementally Learn at Each Single-Hop?","date":"2022-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learn-to-explain-multimodal-reasoning-via","slug":"learn-to-explain-multimodal-reasoning-via","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","date":"2022-09-20","arxiv_id":"2209.09513","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learn-to-explain-multimodal-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2209.09513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.09513"}},"official":{"repos":["lupantech/ScienceQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/can-large-language-models-reason-about","slug":"can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","arxiv_id":"2207.08143","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-reason-about#ran","syntology_url":"https://syntology.ai/paper/2207.08143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08143"}},"official":{"repos":["vlievin/medical-reasoning"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-mostly-know-what-they-know","slug":"language-models-mostly-know-what-they-know","title":"Language Models (Mostly) Know What They Know","date":"2022-07-11","arxiv_id":"2207.05221","repositories_listed":1,"syntology":null},{"url":"/paper/exposing-the-limits-of-video-text-models-1","slug":"exposing-the-limits-of-video-text-models-1","title":"Exposing the Limits of Video-Text Models through Contrast Sets","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/squality-building-a-long-document","slug":"squality-building-a-long-document","title":"SQuALITY: Building a Long-Document Summarization Dataset the Hard Way","date":"2022-05-23","arxiv_id":"2205.11465","repositories_listed":1,"syntology":null},{"url":"/paper/feta-a-benchmark-for-few-sample-task-transfer","slug":"feta-a-benchmark-for-few-sample-task-transfer","title":"FETA: A Benchmark for Few-Sample Task Transfer in Open-Domain Dialogue","date":"2022-05-12","arxiv_id":"2205.06262","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/feta-a-benchmark-for-few-sample-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2205.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.06262"}},"official":{"repos":["alon-albalak/tlidb"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-level-calibration-for-free-form","slug":"answer-level-calibration-for-free-form","title":"Answer-level Calibration for Free-form Multiple Choice Question Answering","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/clues-before-answers-generation-enhanced","slug":"clues-before-answers-generation-enhanced","title":"Clues Before Answers: Generation-Enhanced Multiple-Choice QA","date":"2022-04-30","arxiv_id":"2205.00274","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/clues-before-answers-generation-enhanced#ran","syntology_url":"https://syntology.ai/paper/2205.00274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00274"}},"official":{"repos":["nju-websoft/genmc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-prompts-across-multiple-choice","slug":"evaluating-prompts-across-multiple-choice","title":"Evaluating Prompts Across Multiple Choice Tasks In a Zero-Shot Setting","date":"2022-03-29","arxiv_id":"2203.15754","repositories_listed":1,"syntology":null},{"url":"/paper/medmcqa-a-large-scale-multi-subject-multi","slug":"medmcqa-a-large-scale-multi-subject-multi","title":"MedMCQA : A Large-scale Multi-Subject Multi-Choice Dataset for Medical domain Question Answering","date":"2022-03-27","arxiv_id":"2203.14371","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/medmcqa-a-large-scale-multi-subject-multi#ran","syntology_url":"https://syntology.ai/paper/2203.14371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.14371"}},"official":{"repos":["MedMCQA/MedMCQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/adalogn-adaptive-logic-graph-network-for","slug":"adalogn-adaptive-logic-graph-network-for","title":"AdaLoGN: Adaptive Logic Graph Network for Reasoning-Based Machine Reading Comprehension","date":"2022-03-16","arxiv_id":"2203.08992","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adalogn-adaptive-logic-graph-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08992"}},"official":{"repos":["nju-websoft/adalogn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/all-in-one-exploring-unified-video-language","slug":"all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","arxiv_id":"2203.07303","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-reading-comprehension-questions-2","slug":"what-makes-reading-comprehension-questions-2","title":"What Makes Reading Comprehension Questions Difficult?","date":"2022-03-12","arxiv_id":"2203.06342","repositories_listed":1,"syntology":null},{"url":"/paper/leaf-multiple-choice-question-generation","slug":"leaf-multiple-choice-question-generation","title":"Leaf: Multiple-Choice Question Generation","date":"2022-01-22","arxiv_id":"2201.09012","repositories_listed":1,"syntology":null},{"url":"/paper/multi-choice-questions-based-multi-interest","slug":"multi-choice-questions-based-multi-interest","title":"Multiple Choice Questions based Multi-Interest Policy Learning for Conversational Recommendation","date":"2021-12-22","arxiv_id":"2112.11775","repositories_listed":1,"syntology":null},{"url":"/paper/mirtt-learning-multimodal-interaction","slug":"mirtt-learning-multimodal-interaction","title":"MIRTT: Learning Multimodal Interaction Representations from Trilinear Transformers for Visual Question Answering","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-natural-logic-inference-for","slug":"neural-natural-logic-inference-for","title":"Neural Natural Logic Inference for Interpretable Question Answering","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/surface-form-competition-why-the-highest-1","slug":"surface-form-competition-why-the-highest-1","title":"Surface Form Competition: Why the Highest Probability Answer Isn’t Always Right","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mixqg-neural-question-generation-with-mixed","slug":"mixqg-neural-question-generation-with-mixed","title":"MixQG: Neural Question Generation with Mixed Answer Types","date":"2021-10-15","arxiv_id":"2110.08175","repositories_listed":1,"syntology":null},{"url":"/paper/a-few-more-examples-may-be-worth-billions-of","slug":"a-few-more-examples-may-be-worth-billions-of","title":"A Few More Examples May Be Worth Billions of Parameters","date":"2021-10-08","arxiv_id":"2110.04374","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-few-more-examples-may-be-worth-billions-of#ran","syntology_url":"https://syntology.ai/paper/2110.04374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04374"}},"official":{"repos":["yuvalkirstain/lm-evaluation-harness"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-mrc-framework-for-semantic-role-labeling","slug":"an-mrc-framework-for-semantic-role-labeling","title":"An MRC Framework for Semantic Role Labeling","date":"2021-09-14","arxiv_id":"2109.06660","repositories_listed":1,"syntology":null},{"url":"/paper/arman-pre-training-with-semantically","slug":"arman-pre-training-with-semantically","title":"ARMAN: Pre-training with Semantically Selecting and Reordering of Sentences for Persian Abstractive Summarization","date":"2021-09-09","arxiv_id":"2109.04098","repositories_listed":1,"syntology":null},{"url":"/paper/when-retriever-reader-meets-scenario-based","slug":"when-retriever-reader-meets-scenario-based","title":"When Retriever-Reader Meets Scenario-Based Multiple-Choice Questions","date":"2021-08-31","arxiv_id":"2108.13875","repositories_listed":1,"syntology":null},{"url":"/paper/bert-based-distractor-generation-for-swedish","slug":"bert-based-distractor-generation-for-swedish","title":"BERT-based distractor generation for Swedish reading comprehension questions using a small-scale dataset","date":"2021-08-09","arxiv_id":"2108.03973","repositories_listed":1,"syntology":null},{"url":"/paper/solving-machine-learning-problems","slug":"solving-machine-learning-problems","title":"Solving Machine Learning Problems","date":"2021-07-02","arxiv_id":"2107.01238","repositories_listed":1,"syntology":null},{"url":"/paper/timedial-temporal-commonsense-reasoning-in","slug":"timedial-temporal-commonsense-reasoning-in","title":"TIMEDIAL: Temporal Commonsense Reasoning in Dialog","date":"2021-06-08","arxiv_id":"2106.04571","repositories_listed":1,"syntology":null},{"url":"/paper/prost-physical-reasoning-of-objects-through","slug":"prost-physical-reasoning-of-objects-through","title":"PROST: Physical Reasoning of Objects through Space and Time","date":"2021-06-07","arxiv_id":"2106.03634","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prost-physical-reasoning-of-objects-through#ran","syntology_url":"https://syntology.ai/paper/2106.03634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03634"}},"official":{"repos":["nala-cub/prost"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-ingredients-make-for-an-effective","slug":"what-ingredients-make-for-an-effective","title":"What Ingredients Make for an Effective Crowdsourcing Protocol for Difficult NLU Data Collection Tasks?","date":"2021-06-01","arxiv_id":"2106.00794","repositories_listed":1,"syntology":null},{"url":"/paper/neuralwoz-learning-to-collect-task-oriented","slug":"neuralwoz-learning-to-collect-task-oriented","title":"NeuralWOZ: Learning to Collect Task-Oriented Dialogue via Model-Based Simulation","date":"2021-05-30","arxiv_id":"2105.14454","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neuralwoz-learning-to-collect-task-oriented#ran","syntology_url":"https://syntology.ai/paper/2105.14454","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14454"}},"official":{"repos":["naver-ai/neuralwoz"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/what-to-pre-train-on-efficient-intermediate","slug":"what-to-pre-train-on-efficient-intermediate","title":"What to Pre-Train on? Efficient Intermediate Task Selection","date":"2021-04-16","arxiv_id":"2104.08247","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-to-pre-train-on-efficient-intermediate#ran","syntology_url":"https://syntology.ai/paper/2104.08247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08247"}},"official":{"repos":["adapter-hub/efficient-task-transfer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/explagraphs-an-explanation-graph-generation","slug":"explagraphs-an-explanation-graph-generation","title":"ExplaGraphs: An Explanation Graph Generation Task for Structured Commonsense Reasoning","date":"2021-04-15","arxiv_id":"2104.07644","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/explagraphs-an-explanation-graph-generation#ran","syntology_url":"https://syntology.ai/paper/2104.07644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07644"}},"official":{"repos":["swarnaHub/ExplaGraphs"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fill-in-the-blank-as-a-challenging-video","slug":"fill-in-the-blank-as-a-challenging-video","title":"FIBER: Fill-in-the-Blanks as a Challenging Video Understanding Evaluation Framework","date":"2021-04-09","arxiv_id":"2104.04182","repositories_listed":1,"syntology":null},{"url":"/paper/tsqa-tabular-scenario-based-question","slug":"tsqa-tabular-scenario-based-question","title":"TSQA: Tabular Scenario Based Question Answering","date":"2021-01-14","arxiv_id":"2101.11429","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-nlp-models-via-minimal-contrastive","slug":"explaining-nlp-models-via-minimal-contrastive","title":"Explaining NLP Models via Minimal Contrastive Editing (MiCE)","date":"2020-12-27","arxiv_id":"2012.13985","repositories_listed":1,"syntology":null},{"url":"/paper/option-tracing-beyond-binary-knowledge","slug":"option-tracing-beyond-binary-knowledge","title":"Option Tracing: Beyond Binary Knowledge Tracing","date":"2020-12-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/indicnlpsuite-monolingual-corpora-evaluation","slug":"indicnlpsuite-monolingual-corpora-evaluation","title":"IndicNLPSuite: Monolingual Corpora, Evaluation Benchmarks and Pre-trained Multilingual Language Models for Indian Languages","date":"2020-11-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-bert-based-distractor-generation-scheme-1","slug":"a-bert-based-distractor-generation-scheme-1","title":"A BERT-based Distractor Generation Scheme with Multi-tasking and Negative Answer Training Strategies.","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-bert-based-distractor-generation-scheme","slug":"a-bert-based-distractor-generation-scheme","title":"A BERT-based Distractor Generation Scheme with Multi-tasking and Negative Answer Training Strategies","date":"2020-10-12","arxiv_id":"2010.05384","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-variable-control-for-robust","slug":"counterfactual-variable-control-for-robust","title":"Counterfactual Variable Control for Robust and Interpretable Question Answering","date":"2020-10-12","arxiv_id":"2010.05581","repositories_listed":1,"syntology":null},{"url":"/paper/precise-task-formalization-matters-in","slug":"precise-task-formalization-matters-in","title":"Precise Task Formalization Matters in Winograd Schema Evaluations","date":"2020-10-08","arxiv_id":"2010.04043","repositories_listed":1,"syntology":null},{"url":"/paper/farstail-a-persian-natural-language-inference","slug":"farstail-a-persian-natural-language-inference","title":"FarsTail: A Persian Natural Language Inference Dataset","date":"2020-09-18","arxiv_id":"2009.08820","repositories_listed":1,"syntology":null},{"url":"/paper/fat-albert-finding-answers-in-large-texts","slug":"fat-albert-finding-answers-in-large-texts","title":"FAT ALBERT: Finding Answers in Large Texts using Semantic Similarity Attention Layer based on BERT","date":"2020-08-22","arxiv_id":"2009.01004","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-graph-augmented-abstractive","slug":"knowledge-graph-augmented-abstractive","title":"Knowledge Graph-Augmented Abstractive Summarization with Semantic-Driven Cloze Reward","date":"2020-05-03","arxiv_id":"2005.01159","repositories_listed":1,"syntology":null},{"url":"/paper/lifeqa-a-real-life-dataset-for-video-question","slug":"lifeqa-a-real-life-dataset-for-video-question","title":"LifeQA: A Real-life Dataset for Video Question Answering","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/simulated-annealing-algorithm-for-the","slug":"simulated-annealing-algorithm-for-the","title":"Simulated Annealing Algorithm for the Multiple Choice Multidimensional Knapsack Problem","date":"2020-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/starc-structured-annotations-for-reading","slug":"starc-structured-annotations-for-reading","title":"STARC: Structured Annotations for Reading Comprehension","date":"2020-04-30","arxiv_id":"2004.14797","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/starc-structured-annotations-for-reading#ran","syntology_url":"https://syntology.ai/paper/2004.14797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14797"}},"official":{"repos":["berzak/onestop-qa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/introducing-a-framework-to-assess-newly","slug":"introducing-a-framework-to-assess-newly","title":"Introducing a framework to assess newly created questions with Natural Language Processing","date":"2020-04-28","arxiv_id":"2004.13530","repositories_listed":1,"syntology":null},{"url":"/paper/logic-guided-data-augmentation-and","slug":"logic-guided-data-augmentation-and","title":"Logic-Guided Data Augmentation and Regularization for Consistent Question Answering","date":"2020-04-21","arxiv_id":"2004.10157","repositories_listed":1,"syntology":{"n":18,"n_ran":8,"n_constructed":0,"n_ran_checked":0,"n_instrument":8,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/logic-guided-data-augmentation-and#ran","syntology_url":"https://syntology.ai/paper/2004.10157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.10157"}},"official":{"repos":["AkariAsai/logic_guided_qa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/from-machine-reading-comprehension-to","slug":"from-machine-reading-comprehension-to","title":"From Machine Reading Comprehension to Dialogue State Tracking: Bridging the Gap","date":"2020-04-13","arxiv_id":"2004.05827","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-commonsense-question-answering","slug":"unsupervised-commonsense-question-answering","title":"Unsupervised Commonsense Question Answering with Self-Talk","date":"2020-04-11","arxiv_id":"2004.05483","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unsupervised-commonsense-question-answering#ran","syntology_url":"https://syntology.ai/paper/2004.05483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.05483"}},"official":{"repos":["vered1986/self_talk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/r2de-a-nlp-approach-to-estimating-irt","slug":"r2de-a-nlp-approach-to-estimating-irt","title":"R2DE: a NLP approach to estimating IRT parameters of newly generated questions","date":"2020-01-21","arxiv_id":"2001.07569","repositories_listed":1,"syntology":null},{"url":"/paper/191013291","slug":"191013291","title":"Sentence Embeddings for Russian NLU","date":"2019-10-29","arxiv_id":"1910.13291","repositories_listed":1,"syntology":null},{"url":"/paper/qasc-a-dataset-for-question-answering-via","slug":"qasc-a-dataset-for-question-answering-via","title":"QASC: A Dataset for Question Answering via Sentence Composition","date":"2019-10-25","arxiv_id":"1910.11473","repositories_listed":1,"syntology":null},{"url":"/paper/wiqa-a-dataset-for-what-if-reasoning-over","slug":"wiqa-a-dataset-for-what-if-reasoning-over","title":"WIQA: A dataset for \"What if...\" reasoning over procedural text","date":"2019-09-10","arxiv_id":"1909.04739","repositories_listed":1,"syntology":null},{"url":"/paper/multi-class-hierarchical-question","slug":"multi-class-hierarchical-question","title":"Multi-class Hierarchical Question Classification for Multiple Choice Science Exams","date":"2019-08-15","arxiv_id":"1908.05441","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-english-only-reading-comprehension","slug":"beyond-english-only-reading-comprehension","title":"Beyond English-Only Reading Comprehension: Experiments in Zero-Shot Multilingual Transfer for Bulgarian","date":"2019-08-05","arxiv_id":"1908.01519","repositories_listed":1,"syntology":null},{"url":"/paper/gendered-pronoun-resolution-using-bert-and-an","slug":"gendered-pronoun-resolution-using-bert-and-an","title":"Gendered Pronoun Resolution using BERT and an extractive question answering formulation","date":"2019-06-09","arxiv_id":"1906.03695","repositories_listed":1,"syntology":null},{"url":"/paper/question-answering-as-global-reasoning-over","slug":"question-answering-as-global-reasoning-over","title":"Question Answering as Global Reasoning over Semantic Abstractions","date":"2019-06-09","arxiv_id":"1906.03672","repositories_listed":1,"syntology":null},{"url":"/paper/socialiqa-commonsense-reasoning-about-social","slug":"socialiqa-commonsense-reasoning-about-social","title":"SocialIQA: Commonsense Reasoning about Social Interactions","date":"2019-04-22","arxiv_id":"1904.09728","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/socialiqa-commonsense-reasoning-about-social#ran","syntology_url":"https://syntology.ai/paper/1904.09728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09728"}},"official":null}},{"url":"/paper/probing-prior-knowledge-needed-in-challenging","slug":"probing-prior-knowledge-needed-in-challenging","title":"Investigating Prior Knowledge for Challenging Chinese Machine Reading Comprehension","date":"2019-04-21","arxiv_id":"1904.09679","repositories_listed":1,"syntology":null},{"url":"/paper/eliminet-a-model-for-eliminating-options-for-1","slug":"eliminet-a-model-for-eliminating-options-for-1","title":"ElimiNet: A Model for Eliminating Options for Reading Comprehension with Multiple Choice Questions","date":"2019-04-04","arxiv_id":"1904.02651","repositories_listed":1,"syntology":null},{"url":"/paper/evidence-sentence-extraction-for-machine","slug":"evidence-sentence-extraction-for-machine","title":"Evidence Sentence Extraction for Machine Reading Comprehension","date":"2019-02-23","arxiv_id":"1902.08852","repositories_listed":1,"syntology":null},{"url":"/paper/improving-question-answering-with-external","slug":"improving-question-answering-with-external","title":"Improving Question Answering with External Knowledge","date":"2019-02-03","arxiv_id":"1902.00993","repositories_listed":1,"syntology":null},{"url":"/paper/dream-a-challenge-dataset-and-models-for","slug":"dream-a-challenge-dataset-and-models-for","title":"DREAM: A Challenge Dataset and Models for Dialogue-Based Reading Comprehension","date":"2019-02-01","arxiv_id":"1902.00164","repositories_listed":1,"syntology":null},{"url":"/paper/video-prediction-via-selective-sampling","slug":"video-prediction-via-selective-sampling","title":"Video Prediction via Selective Sampling","date":"2018-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-machine-reading-comprehension-with","slug":"improving-machine-reading-comprehension-with","title":"Improving Machine Reading Comprehension with General Reading Strategies","date":"2018-10-31","arxiv_id":"1810.13441","repositories_listed":1,"syntology":null},{"url":"/paper/extracting-keywords-from-open-ended-business","slug":"extracting-keywords-from-open-ended-business","title":"Extracting Keywords from Open-Ended Business Survey Questions","date":"2018-08-31","arxiv_id":"1808.10685","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-reading-comprehension-questions","slug":"what-makes-reading-comprehension-questions","title":"What Makes Reading Comprehension Questions Easier?","date":"2018-08-28","arxiv_id":"1808.09384","repositories_listed":1,"syntology":null},{"url":"/paper/cnn-for-text-based-multiple-choice-question","slug":"cnn-for-text-based-multiple-choice-question","title":"CNN for Text-Based Multiple Choice Question Answering","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/distractor-generation-for-multiple-choice","slug":"distractor-generation-for-multiple-choice","title":"Distractor Generation for Multiple Choice Questions Using Learning to Rank","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/constructing-narrative-event-evolutionary","slug":"constructing-narrative-event-evolutionary","title":"Constructing Narrative Event Evolutionary Graph for Script Event Prediction","date":"2018-05-14","arxiv_id":"1805.05081","repositories_listed":1,"syntology":null},{"url":"/paper/structured-triplet-learning-with-pos-tag","slug":"structured-triplet-learning-with-pos-tag","title":"Structured Triplet Learning with POS-tag Guided Attention for Visual Question Answering","date":"2018-01-24","arxiv_id":"1801.07853","repositories_listed":1,"syntology":null},{"url":"/paper/vqs-linking-segmentations-to-questions-and","slug":"vqs-linking-segmentations-to-questions-and","title":"VQS: Linking Segmentations to Questions and Answers for Supervised Attention in VQA and Question-Focused Semantic Segmentation","date":"2017-08-15","arxiv_id":"1708.04686","repositories_listed":1,"syntology":null},{"url":"/paper/which-is-the-effective-way-for-gaokao","slug":"which-is-the-effective-way-for-gaokao","title":"Which is the Effective Way for Gaokao: Information Retrieval or Neural Networks?","date":"2017-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-residual-learning-for-visual-qa","slug":"multimodal-residual-learning-for-visual-qa","title":"Multimodal Residual Learning for Visual QA","date":"2016-06-05","arxiv_id":"1606.01455","repositories_listed":1,"syntology":null},{"url":"/paper/joint-learning-of-sentence-embeddings-for","slug":"joint-learning-of-sentence-embeddings-for","title":"Joint Learning of Sentence Embeddings for Relevance and Entailment","date":"2016-05-16","arxiv_id":"1605.04655","repositories_listed":1,"syntology":null},{"url":null,"slug":"hats-hindi-analogy-test-set-for-evaluating","title":"HATS: Hindi Analogy Test Set for Evaluating Reasoning in Large Language Models","date":"2025-07-17","arxiv_id":"2507.13238","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-generative-energy-arena-gea-incorporating","title":"The Generative Energy Arena (GEA): Incorporating Energy Awareness in Large Language Model (LLM) Human Evaluations","date":"2025-07-17","arxiv_id":"2507.13302","repositories_listed":0,"syntology":null},{"url":null,"slug":"mateinfoub-a-real-world-benchmark-for-testing","title":"MateInfoUB: A Real-World Benchmark for Testing LLMs in Competitive, Multilingual, and Multimodal Educational Tasks","date":"2025-07-03","arxiv_id":"2507.03162","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-financial-reasoning-at-scale-a","title":"Advanced Financial Reasoning at Scale: A Comprehensive Evaluation of Large Language Models on CFA Level III","date":"2025-06-29","arxiv_id":"2507.02954","repositories_listed":0,"syntology":null},{"url":null,"slug":"omnieval-a-benchmark-for-evaluating-omni","title":"OmniEval: A Benchmark for Evaluating Omni-modal Models with Visual, Auditory, and Textual Inputs","date":"2025-06-26","arxiv_id":"2506.20960","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-vision-language-models-for","title":"Adapting Vision-Language Models for Evaluating World Models","date":"2025-06-22","arxiv_id":"2506.17967","repositories_listed":0,"syntology":null},{"url":null,"slug":"physunibench-an-undergraduate-level-physics","title":"PhysUniBench: An Undergraduate-Level Physics Reasoning Benchmark for Multimodal Models","date":"2025-06-21","arxiv_id":"2506.17667","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-far-can-off-the-shelf-multimodal-large","title":"How Far Can Off-the-Shelf Multimodal Large Language Models Go in Online Episodic Memory Question Answering?","date":"2025-06-19","arxiv_id":"2506.16450","repositories_listed":0,"syntology":null},{"url":null,"slug":"wikimixqa-a-multimodal-benchmark-for-question","title":"WikiMixQA: A Multimodal Benchmark for Question Answering over Tables and Charts","date":"2025-06-18","arxiv_id":"2506.15594","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypothesis-testing-for-quantifying-llm-human","title":"Hypothesis Testing for Quantifying LLM-Human Misalignment in Multiple Choice Settings","date":"2025-06-17","arxiv_id":"2506.14997","repositories_listed":0,"syntology":null},{"url":"/paper/thunder-nubench-a-benchmark-for-llms-sentence","slug":"thunder-nubench-a-benchmark-for-llms-sentence","title":"Thunder-NUBench: A Benchmark for LLMs' Sentence-Level Negation Understanding","date":"2025-06-17","arxiv_id":"2506.14397","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-tuning-and-cot-prompting-for","title":"Instruction Tuning and CoT Prompting for Contextual Medical QA with LLMs","date":"2025-06-13","arxiv_id":"2506.12182","repositories_listed":0,"syntology":null},{"url":null,"slug":"different-questions-different-models-fine","title":"Different Questions, Different Models: Fine-Grained Evaluation of Uncertainty and Calibration in Clinical QA with LLMs","date":"2025-06-12","arxiv_id":"2506.10769","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-shortcut-aware-video-qa-benchmark-for","title":"A Shortcut-aware Video-QA Benchmark for Physical Understanding via Minimal Video Pairs","date":"2025-06-11","arxiv_id":"2506.09987","repositories_listed":0,"syntology":null},{"url":null,"slug":"versavid-r1-a-versatile-video-understanding","title":"VersaVid-R1: A Versatile Video Understanding and Reasoning Model from Question Answering to Captioning Tasks","date":"2025-06-10","arxiv_id":"2506.09079","repositories_listed":0,"syntology":null},{"url":null,"slug":"argus-hallucination-and-omission-evaluation","title":"ARGUS: Hallucination and Omission Evaluation in Video-LLMs","date":"2025-06-09","arxiv_id":"2506.07371","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-llm-corrupted-crowdsourcing-data","title":"Evaluating LLM-corrupted Crowdsourcing Data Without Ground Truth","date":"2025-06-08","arxiv_id":"2506.06991","repositories_listed":0,"syntology":null}],"record_sha256":"03b9f689b3745c736974371a730997b074ebb8f024837c2ae0ecb8fb26247ab6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}