{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/9","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":12,"rows_per_page":100,"rows":[801,900],"of":1107,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice","prev":"/task/multiple-choice/papers/8","next":"/task/multiple-choice/papers/10","papers":[{"url":null,"slug":"kornat-llm-alignment-benchmark-for-korean","title":"KorNAT: LLM Alignment Benchmark for Korean Social Values and Common Knowledge","date":"2024-02-21","arxiv_id":"2402.13605","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranking-large-language-models-without-ground","title":"Ranking Large Language Models without Ground Truth","date":"2024-02-21","arxiv_id":"2402.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-comprehensibility-assessment-of","title":"Digital Comprehensibility Assessment of Simplified Texts among Persons with Intellectual Disabilities","date":"2024-02-20","arxiv_id":"2402.13094","repositories_listed":0,"syntology":null},{"url":null,"slug":"stick-to-your-role-stability-of-personal","title":"Stick to your Role! Stability of Personal Values Expressed in Large Language Models","date":"2024-02-19","arxiv_id":"2402.14846","repositories_listed":0,"syntology":null},{"url":null,"slug":"kmmlu-measuring-massive-multitask-language","title":"KMMLU: Measuring Massive Multitask Language Understanding in Korean","date":"2024-02-18","arxiv_id":"2402.11548","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-implicit-discourse-relation","title":"Prompting Implicit Discourse Relation Annotation","date":"2024-02-07","arxiv_id":"2402.04918","repositories_listed":0,"syntology":null},{"url":null,"slug":"minds-versus-machines-rethinking-entailment","title":"Are Machines Better at Complex Reasoning? Unveiling Human-Machine Inference Gaps in Entailment Verification","date":"2024-02-06","arxiv_id":"2402.03686","repositories_listed":0,"syntology":null},{"url":null,"slug":"scemqa-a-scientific-college-entrance-level","title":"SceMQA: A Scientific College Entrance Level Multimodal Question Answering Benchmark","date":"2024-02-06","arxiv_id":"2402.05138","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-answers-reviewing-the-rationality","title":"LLMs May Perform MCQA by Selecting the Least Incorrect Option","date":"2024-02-02","arxiv_id":"2402.01349","repositories_listed":0,"syntology":null},{"url":null,"slug":"distractor-generation-for-multiple-choice-2","title":"Distractor Generation in Multiple-Choice Tasks: A Survey of Methods, Datasets, and Evaluation","date":"2024-02-02","arxiv_id":"2402.01512","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-llm-generated-multimodal-diagnosis","title":"Evaluating LLM -- Generated Multimodal Diagnosis from Medical Images and Symptom Analysis","date":"2024-01-28","arxiv_id":"2402.01730","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-collective-superintelligence","title":"Towards Collective Superintelligence: Amplifying Group IQ using Conversational Swarms","date":"2024-01-25","arxiv_id":"2401.15109","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-weight-experiments-for-llm-instruction","title":"Instruction Fine-Tuning: Does Prompt Loss Matter?","date":"2024-01-24","arxiv_id":"2401.13586","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-calibration-gap-between-model-and-human","title":"What Large Language Models Know and What People Think They Know","date":"2024-01-24","arxiv_id":"2401.13835","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-answer-validation-using-text","title":"Automated Answer Validation using Text Similarity","date":"2024-01-13","arxiv_id":"2401.08688","repositories_listed":0,"syntology":null},{"url":null,"slug":"pub-a-pragmatics-understanding-benchmark-for","title":"PUB: A Pragmatics Understanding Benchmark for Assessing LLMs' Pragmatics Capabilities","date":"2024-01-13","arxiv_id":"2401.07078","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-joint-reasoning-based-disease-q-a-system","title":"A Joint-Reasoning based Disease Q&A System","date":"2024-01-06","arxiv_id":"2401.03181","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-earth-is-flat-unveiling-factual-errors-in","title":"The Earth is Flat? Unveiling Factual Errors in Large Language Models","date":"2024-01-01","arxiv_id":"2401.00761","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusionmind-improving-question-and-answering","title":"FusionMind -- Improving question and answering with external context fusion","date":"2023-12-31","arxiv_id":"2401.00388","repositories_listed":0,"syntology":null},{"url":null,"slug":"bloomvqa-assessing-hierarchical-multi-modal","title":"BloomVQA: Assessing Hierarchical Multi-modal Comprehension","date":"2023-12-20","arxiv_id":"2312.12716","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2023-a-summary-of-the-first","title":"Perception Test 2023: A Summary of the First Challenge And Outcome","date":"2023-12-20","arxiv_id":"2312.13090","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-evaluation-improves-selective-generation","title":"Self-Evaluation Improves Selective Generation in Large Language Models","date":"2023-12-14","arxiv_id":"2312.09300","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-foundational-multimodal-vision-language-ai","title":"A Foundational Multimodal Vision Language AI Assistant for Human Pathology","date":"2023-12-13","arxiv_id":"2312.07814","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-ai-generated-gpt-4-and","title":"A Comparative Study of AI-Generated (GPT-4) and Human-crafted MCQs in Programming Education","date":"2023-12-05","arxiv_id":"2312.03173","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-the-potential-of-large-language","title":"Unleashing the Potential of Large Language Model: Zero-shot VQA for Flood Disaster Scenario","date":"2023-12-04","arxiv_id":"2312.01882","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-rationale-understanding-of","title":"Evaluating the Rationale Understanding of Critical Reasoning in Logical Reading Comprehension","date":"2023-11-30","arxiv_id":"2311.18353","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-data-contamination-in-modern","title":"Investigating Data Contamination in Modern Benchmarks for Large Language Models","date":"2023-11-16","arxiv_id":"2311.09783","repositories_listed":0,"syntology":null},{"url":null,"slug":"psybench-a-balanced-and-in-depth","title":"ConceptPsy:A Benchmark Suite with Conceptual Comprehensiveness in Psychology","date":"2023-11-16","arxiv_id":"2311.09861","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-llms-on-document-based-qa-exact","title":"Evaluating LLMs on Document-Based QA: Exact Answer Selection and Numerical Extraction using Cogtale dataset","date":"2023-11-14","arxiv_id":"2311.07878","repositories_listed":0,"syntology":null},{"url":null,"slug":"characterizing-large-language-models-as","title":"Characterizing Large Language Models as Rationalizers of Knowledge-intensive Tasks","date":"2023-11-09","arxiv_id":"2311.05085","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-distractors-in-multiple-choice","title":"Assessing Distractors in Multiple-Choice Tests","date":"2023-11-08","arxiv_id":"2311.04554","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-multiple-large-language-models-in","title":"Evaluating multiple large language models in pediatric ophthalmology","date":"2023-11-07","arxiv_id":"2311.04368","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-potential-of-leading-large","title":"Evaluating the Potential of Leading Large Language Models in Reasoning Biology Questions","date":"2023-11-05","arxiv_id":"2311.07582","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-robots-are-coming-large-multimodal","title":"More Robots are Coming: Large Multimodal Models (ChatGPT) can Solve Visually Diverse Images of Parsons Problems","date":"2023-11-03","arxiv_id":"2311.04926","repositories_listed":0,"syntology":null},{"url":null,"slug":"desiq-towards-an-unbiased-challenging","title":"DeSIQ: Towards an Unbiased, Challenging Benchmark for Social Intelligence Understanding","date":"2023-10-24","arxiv_id":"2310.18359","repositories_listed":0,"syntology":null},{"url":null,"slug":"dataset-bias-mitigation-in-multiple-choice","title":"Dataset Bias Mitigation in Multiple-Choice Visual Question Answering and Beyond","date":"2023-10-23","arxiv_id":"2310.14670","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-symbol-binding-ability-of","title":"Evaluating the Symbol Binding Ability of Large Language Models for Multiple-Choice Questions in Vietnamese General Education","date":"2023-10-18","arxiv_id":"2310.12059","repositories_listed":0,"syntology":null},{"url":null,"slug":"field-testing-items-using-artificial","title":"Field-testing items using artificial intelligence: Natural language processing with transformers","date":"2023-10-18","arxiv_id":"2310.11655","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-uncertainty-calibration-of","title":"Investigating Uncertainty Calibration of Aligned Language Models under the Multiple-Choice Setting","date":"2023-10-18","arxiv_id":"2310.11732","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-bias-for-question-answering-models","title":"Mitigating Bias for Question Answering Models by Tracking Bias Influence","date":"2023-10-13","arxiv_id":"2310.08795","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-performance-of-multimodal-language","title":"On the Performance of Multimodal Language Models","date":"2023-10-04","arxiv_id":"2310.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-question-generation-from","title":"Automating question generation from educational text","date":"2023-09-26","arxiv_id":"2309.15004","repositories_listed":0,"syntology":null},{"url":null,"slug":"hans-are-you-clever-clever-hans-effect","title":"HANS, are you clever? Clever Hans Effect Analysis of Neural Systems","date":"2023-09-21","arxiv_id":"2309.12481","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarks-for-pira-2-0-a-reading","title":"Benchmarks for Pirá 2.0, a Reading Comprehension Dataset about the Ocean, the Brazilian Coast, and Climate Change","date":"2023-09-19","arxiv_id":"2309.10945","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-models-are-susceptible-to-incorrect","title":"Language models are susceptible to incorrect patient self-diagnosis in medical applications","date":"2023-09-17","arxiv_id":"2309.09362","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-applicability-of-self","title":"Self-Assessment Tests are Unreliable Measures of LLM Personality","date":"2023-09-15","arxiv_id":"2309.08163","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-of-chatgpt-3-5-and-gpt-4-on-the","title":"Performance of ChatGPT-3.5 and GPT-4 on the United States Medical Licensing Examination With and Without Distractions","date":"2023-09-12","arxiv_id":"2309.08625","repositories_listed":0,"syntology":null},{"url":null,"slug":"use-neural-networks-to-recognize-students","title":"Use neural networks to recognize students' handwritten letters and incorrect symbols","date":"2023-09-12","arxiv_id":"2309.06221","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-automatic-evaluation-framework-for-multi","title":"An Automatic Evaluation Framework for Multi-turn Medical Consultations Capabilities of Large Language Models","date":"2023-09-05","arxiv_id":"2309.02077","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalised-winograd-schema-and-its","title":"Generalised Winograd Schema and its Contextuality","date":"2023-08-31","arxiv_id":"2308.16498","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-sensitivity-to-the","title":"Large Language Models Sensitivity to The Order of Options in Multiple-Choice Questions","date":"2023-08-22","arxiv_id":"2308.11483","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-open-source-large","title":"A Comparative Study of Open-Source Large Language Models, GPT-4 and Claude 2: Multiple-Choice Test Taking in Nephrology","date":"2023-08-09","arxiv_id":"2308.04709","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-circuit-analysis-interpretability-scale","title":"Does Circuit Analysis Interpretability Scale? Evidence from Multiple Choice Capabilities in Chinchilla","date":"2023-07-18","arxiv_id":"2307.09458","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-multiple-choice-reading-and","title":"Analyzing Multiple-Choice Reading and Listening Comprehension Tests","date":"2023-07-03","arxiv_id":"2307.01076","repositories_listed":0,"syntology":null},{"url":null,"slug":"camchoice-a-corpus-of-multiple-choice","title":"Analysis of the Cambridge Multiple-Choice Questions Reading Dataset with a Focus on Candidate Response Distribution","date":"2023-06-22","arxiv_id":"2306.13047","repositories_listed":0,"syntology":null},{"url":null,"slug":"recap-kg-mining-knowledge-graphs-from-raw-gp","title":"RECAP-KG: Mining Knowledge Graphs from Raw GP Notes for Remote COVID-19 Assessment in Primary Care","date":"2023-06-17","arxiv_id":"2306.17175","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-chatgpt-pass-the-vietnamese-national-high","title":"Can ChatGPT pass the Vietnamese National High School Graduation Examination?","date":"2023-06-15","arxiv_id":"2306.09170","repositories_listed":0,"syntology":null},{"url":null,"slug":"thrilled-by-your-progress-large-language","title":"Thrilled by Your Progress! Large Language Models (GPT-4) No Longer Struggle to Pass Assessments in Higher Education Programming Courses","date":"2023-06-15","arxiv_id":"2306.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-effectiveness-of-chatgpt-in","title":"Investigating the Effectiveness of ChatGPT in Mathematical Reasoning and Problem Solving: Evidence from the Vietnamese National High School Graduation Examination","date":"2023-06-10","arxiv_id":"2306.06331","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-based-representations-and-dynamic","title":"Network-based Representations and Dynamic Discrete Choice Models for Multiple Discrete Choice Analysis","date":"2023-06-07","arxiv_id":"2306.04606","repositories_listed":0,"syntology":null},{"url":null,"slug":"buca-a-binary-classification-approach-to","title":"BUCA: A Binary Classification Approach to Unsupervised Commonsense Question Answering","date":"2023-05-25","arxiv_id":"2305.15932","repositories_listed":0,"syntology":null},{"url":null,"slug":"have-large-language-models-developed-a","title":"Have Large Language Models Developed a Personality?: Applicability of Self-Assessment Tests in Measuring Personality in LLMs","date":"2023-05-24","arxiv_id":"2305.14693","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-a-choice-knowledge-base-question","title":"Make a Choice! Knowledge Base Question Answering with In-Context Learning","date":"2023-05-23","arxiv_id":"2305.13972","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-rewriting-for-retrieval-augmented-large","title":"Query Rewriting for Retrieval-Augmented Large Language Models","date":"2023-05-23","arxiv_id":"2305.14283","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-response-interpretation-for","title":"Contextual Response Interpretation for Automated Structured Interviews: A Case Study in Market Research","date":"2023-04-30","arxiv_id":"2305.00577","repositories_listed":0,"syntology":null},{"url":null,"slug":"who-s-the-best-detective-llms-vs-mls-in","title":"Who's the Best Detective? LLMs vs. MLs in Detecting Incoherent Fourth Grade Math Answers","date":"2023-04-21","arxiv_id":"2304.11257","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-performance-of-chatgpt-in","title":"Analyzing the Performance of ChatGPT in Cardiology and Vascular Pathologies","date":"2023-04-15","arxiv_id":"2307.02518","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-engineering-and-calibration-for-zero","title":"Prompt Engineering and Calibration for Zero-Shot Commonsense Reasoning","date":"2023-04-14","arxiv_id":"2304.06962","repositories_listed":0,"syntology":null},{"url":null,"slug":"disto-evaluating-textual-distractors-for","title":"DISTO: Evaluating Textual Distractors for Multi-Choice Questions using Negative Sampling based Approach","date":"2023-04-10","arxiv_id":"2304.04881","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-language-gap-knowledge-injected","title":"Bridging the Language Gap: Knowledge Injected Multilingual Question Answering","date":"2023-04-06","arxiv_id":"2304.03159","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-4-to-gpt-3-5-hold-my-scalpel-a-look-at","title":"GPT-4 to GPT-3.5: 'Hold My Scalpel' -- A Look at the Competency of OpenAI's GPT on the Plastic Surgery In-Service Training Exam","date":"2023-04-04","arxiv_id":"2304.01503","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-generation-of-multiple-choice","title":"Automatic Generation of Multiple-Choice Questions","date":"2023-03-25","arxiv_id":"2303.14576","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-guided-reasoning-approach-for-open","title":"A Graph-Guided Reasoning Approach for Open-ended Commonsense Question Answering","date":"2023-03-18","arxiv_id":"2303.10395","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-generative-pre-trained-transformers-gpt","title":"Can Generative Pre-trained Transformers (GPT) Pass Assessments in Higher Education Programming Courses?","date":"2023-03-16","arxiv_id":"2303.09325","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-multiple-choice-questions-for","title":"Generating multiple-choice questions for medical question answering with distractors and cue-masking","date":"2023-03-13","arxiv_id":"2303.07069","repositories_listed":0,"syntology":null},{"url":"/paper/multi-efficient-video-and-language","slug":"multi-efficient-video-and-language","title":"MuLTI: Efficient Video-and-Language Understanding with Text-Guided MultiWay-Sampler and Multiple Choice Modeling","date":"2023-03-10","arxiv_id":"2303.05707","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-gpt-struggle-to-answer","title":"Large Language Models (GPT) Struggle to Answer Multiple-Choice Questions about Code","date":"2023-03-09","arxiv_id":"2303.08033","repositories_listed":0,"syntology":null},{"url":null,"slug":"tess-zero-shot-classification-via-textual","title":"Empowering Sentence Encoders with Prompting and Label Retrieval for Zero-shot Text Classification","date":"2022-12-20","arxiv_id":"2212.10391","repositories_listed":0,"syntology":null},{"url":null,"slug":"true-detective-a-challenging-benchmark-for","title":"True Detective: A Deep Abductive Reasoning Benchmark Undoable for GPT-3 and Challenging for GPT-4","date":"2022-12-20","arxiv_id":"2212.10114","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-type-identification-for-academic","title":"Question-type Identification for Academic Questions in Online Learning Platform","date":"2022-11-24","arxiv_id":"2211.13727","repositories_listed":0,"syntology":null},{"url":null,"slug":"agree-a-system-for-generating-automated","title":"AGReE: A system for generating Automated Grammar Reading Exercises","date":"2022-10-28","arxiv_id":"2210.16302","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-based-arabic-language-and-speech-tutor","title":"AI-based Arabic Language and Speech Tutor","date":"2022-10-22","arxiv_id":"2210.12346","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-turn-debate-doesn-t-help-humans-answer","title":"Two-Turn Debate Doesn't Help Humans Answer Hard Reading Comprehension Questions","date":"2022-10-19","arxiv_id":"2210.10860","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-prior-bias-and-choice-paralysis","title":"Understanding Prior Bias and Choice Paralysis in Transformer-based Language Representation Models through Four Experimental Probes","date":"2022-10-03","arxiv_id":"2210.01258","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-weak-supervision-approach-for-predicting","title":"A Weak Supervision Approach for Predicting Difficulty of Technical Interview Questions","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"document-level-event-factuality-1","title":"Document-level Event Factuality Identification via Machine Reading Comprehension Frameworks with Transfer Learning","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-contradictions-to-improve-qa-systems","title":"Using contradictions improves question answering systems","date":"2022-09-28","arxiv_id":"2211.05598","repositories_listed":0,"syntology":null},{"url":null,"slug":"identification-of-the-marginal-treatment","title":"Treatment Effects with Multidimensional Unobserved Heterogeneity: Identification of the Marginal Treatment Effect","date":"2022-09-23","arxiv_id":"2209.11444","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-choice-question-generation-towards","title":"Multiple-Choice Question Generation: Towards an Automated Assessment Framework","date":"2022-09-23","arxiv_id":"2209.11830","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-algorithms-for-federated-learning","title":"Scheduling Algorithms for Federated Learning with Minimal Energy Consumption","date":"2022-09-13","arxiv_id":"2209.06210","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-event-causality-identification-with","title":"Zero-shot Event Causality Identification with Question Answering","date":"2022-09-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dataset-and-benchmark-for-automatically","title":"From Human Days to Machine Seconds: Automatically Answering and Generating Machine Learning Final Exams","date":"2022-06-11","arxiv_id":"2206.05442","repositories_listed":0,"syntology":null},{"url":null,"slug":"croatpas-a-survey-based-evaluation","title":"CroaTPAS: A Survey-based Evaluation","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hrca-advanced-multiple-choice-machine-reading","title":"HRCA+: Advanced Multiple-choice Machine Reading Comprehension Method","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"paddle-a-platform-to-identify-complex-words","title":"PADDLe: a Platform to Identify Complex Words for Learners of French as a Foreign Language (FFL)","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/answer-uncertainty-and-unanswerability-in-1","slug":"answer-uncertainty-and-unanswerability-in-1","title":"Answer Uncertainty and Unanswerability in Multiple-Choice Machine Reading Comprehension","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-generation-of-distractors-for-fill","title":"Automatic Generation of Distractors for Fill-in-the-Blank Exercises with Round-Trip Neural Machine Translation","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"clozer-adaptable-data-augmentation-for-cloze-1","title":"Clozer”:\" Adaptable Data Augmentation for Cloze-style Reading Comprehension","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-multiple-choice-question-1","title":"Unsupervised multiple-choice question generation for out-of-domain Q&A fine-tuning","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"34da0af5e302405ff373170bc4f134346c4489ebf4e67e1381f0ed6fdc71c5e8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}