{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reading-comprehension/papers/2","list_of":"/task/reading-comprehension","task":"Reading Comprehension","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":18,"rows_per_page":100,"rows":[101,200],"of":1760,"counts":{"archive_papers_tagged":1760,"with_a_code_link":634,"where_syntology_ran_a_sample":139,"not_listed_spam_title":0,"listed":1760,"listed_where_code_ran":139,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":120,"every_run_a_failure_of_syntologys_instrument":19,"listed_with_a_run_with_no_instrument_failure":120,"listed_every_run_a_failure_of_syntologys_instrument":19,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reading-comprehension","prev":"/task/reading-comprehension","next":"/task/reading-comprehension/papers/3","papers":[{"url":"/paper/numnet-machine-reading-comprehension-with","slug":"numnet-machine-reading-comprehension-with","title":"NumNet: Machine Reading Comprehension with Numerical Reasoning","date":"2019-10-15","arxiv_id":"1910.06701","repositories_listed":2,"syntology":null},{"url":"/paper/mmm-multi-stage-multi-task-learning-for-multi","slug":"mmm-multi-stage-multi-task-learning-for-multi","title":"MMM: Multi-stage Multi-task Learning for Multi-choice Reading Comprehension","date":"2019-10-01","arxiv_id":"1910.00458","repositories_listed":2,"syntology":null},{"url":"/paper/revealing-the-importance-of-semantic","slug":"revealing-the-importance-of-semantic","title":"Revealing the Importance of Semantic Retrieval for Machine Reading at Scale","date":"2019-09-17","arxiv_id":"1909.08041","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":2,"n_honours":4,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 4 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revealing-the-importance-of-semantic#ran","syntology_url":"https://syntology.ai/paper/1909.08041","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08041"}},"official":{"repos":["easonnie/semanticRetrievalMRS"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dcmn-dual-co-matching-network-for-multi","slug":"dcmn-dual-co-matching-network-for-multi","title":"DCMN+: Dual Co-Matching Network for Multi-choice Reading Comprehension","date":"2019-08-30","arxiv_id":"1908.11511","repositories_listed":2,"syntology":null},{"url":"/paper/pre-training-with-whole-word-masking-for","slug":"pre-training-with-whole-word-masking-for","title":"Pre-Training with Whole Word Masking for Chinese BERT","date":"2019-06-19","arxiv_id":"1906.08101","repositories_listed":2,"syntology":null},{"url":"/paper/multi-hop-reading-comprehension-through","slug":"multi-hop-reading-comprehension-through","title":"Multi-hop Reading Comprehension through Question Decomposition and Rescoring","date":"2019-06-07","arxiv_id":"1906.02916","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-hop-reading-comprehension-through#ran","syntology_url":"https://syntology.ai/paper/1906.02916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.02916"}},"official":{"repos":["shmsw25/DecompRC"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-question-answer-hierarchies","slug":"generating-question-answer-hierarchies","title":"Generating Question-Answer Hierarchies","date":"2019-06-06","arxiv_id":"1906.02622","repositories_listed":2,"syntology":null},{"url":"/paper/190600318","slug":"190600318","title":"Question Answering as an Automatic Evaluation Metric for News Article Summarization","date":"2019-06-02","arxiv_id":"1906.00318","repositories_listed":2,"syntology":null},{"url":"/paper/adaptation-of-deep-bidirectional-multilingual","slug":"adaptation-of-deep-bidirectional-multilingual","title":"Adaptation of Deep Bidirectional Multilingual Transformers for Russian Language","date":"2019-05-17","arxiv_id":"1905.07213","repositories_listed":2,"syntology":null},{"url":"/paper/cognitive-graph-for-multi-hop-reading","slug":"cognitive-graph-for-multi-hop-reading","title":"Cognitive Graph for Multi-Hop Reading Comprehension at Scale","date":"2019-05-14","arxiv_id":"1905.05460","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cognitive-graph-for-multi-hop-reading#ran","syntology_url":"https://syntology.ai/paper/1905.05460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05460"}},"official":{"repos":["THUDM/CogQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/fastfusionnet-new-state-of-the-art-for","slug":"fastfusionnet-new-state-of-the-art-for","title":"FastFusionNet: New State-of-the-Art for DAWNBench SQuAD","date":"2019-02-28","arxiv_id":"1902.11291","repositories_listed":2,"syntology":null},{"url":"/paper/densely-connected-attention-propagation-for","slug":"densely-connected-attention-propagation-for","title":"Densely Connected Attention Propagation for Reading Comprehension","date":"2018-11-10","arxiv_id":"1811.04210","repositories_listed":2,"syntology":null},{"url":"/paper/commonsense-for-generative-multi-hop-question","slug":"commonsense-for-generative-multi-hop-question","title":"Commonsense for Generative Multi-Hop Question Answering Tasks","date":"2018-09-17","arxiv_id":"1809.06309","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/commonsense-for-generative-multi-hop-question#ran","syntology_url":"https://syntology.ai/paper/1809.06309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06309"}},"official":{"repos":["yicheng-w/CommonSenseMultiHopQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/generating-distractors-for-reading","slug":"generating-distractors-for-reading","title":"Generating Distractors for Reading Comprehension Questions from Real Examinations","date":"2018-09-08","arxiv_id":"1809.02768","repositories_listed":2,"syntology":null},{"url":"/paper/jack-the-reader-a-machine-reading-framework","slug":"jack-the-reader-a-machine-reading-framework","title":"Jack the Reader - A Machine Reading Framework","date":"2018-06-20","arxiv_id":"1806.08727","repositories_listed":2,"syntology":null},{"url":"/paper/allennlp-a-deep-semantic-natural-language","slug":"allennlp-a-deep-semantic-natural-language","title":"AllenNLP: A Deep Semantic Natural Language Processing Platform","date":"2018-03-20","arxiv_id":"1803.07640","repositories_listed":2,"syntology":null},{"url":"/paper/the-web-as-a-knowledge-base-for-answering","slug":"the-web-as-a-knowledge-base-for-answering","title":"The Web as a Knowledge-base for Answering Complex Questions","date":"2018-03-18","arxiv_id":"1803.06643","repositories_listed":2,"syntology":null},{"url":"/paper/yuanfudao-at-semeval-2018-task-11-three-way","slug":"yuanfudao-at-semeval-2018-task-11-three-way","title":"Yuanfudao at SemEval-2018 Task 11: Three-way Attention and Relational Knowledge for Commonsense Machine Comprehension","date":"2018-03-01","arxiv_id":"1803.00191","repositories_listed":2,"syntology":null},{"url":"/paper/bidirectional-attention-for-sql-generation","slug":"bidirectional-attention-for-sql-generation","title":"Bidirectional Attention for SQL Generation","date":"2017-12-30","arxiv_id":"1801.00076","repositories_listed":2,"syntology":null},{"url":"/paper/the-narrativeqa-reading-comprehension","slug":"the-narrativeqa-reading-comprehension","title":"The NarrativeQA Reading Comprehension Challenge","date":"2017-12-19","arxiv_id":"1712.07040","repositories_listed":2,"syntology":null},{"url":"/paper/fast-reading-comprehension-with-convnets","slug":"fast-reading-comprehension-with-convnets","title":"Fast Reading Comprehension with ConvNets","date":"2017-11-12","arxiv_id":"1711.04352","repositories_listed":2,"syntology":null},{"url":"/paper/two-stage-synthesis-networks-for-transfer","slug":"two-stage-synthesis-networks-for-transfer","title":"Two-Stage Synthesis Networks for Transfer Learning in Machine Comprehension","date":"2017-06-29","arxiv_id":"1706.09789","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-relation-extraction-via-reading","slug":"zero-shot-relation-extraction-via-reading","title":"Zero-Shot Relation Extraction via Reading Comprehension","date":"2017-06-13","arxiv_id":"1706.04115","repositories_listed":2,"syntology":null},{"url":"/paper/race-large-scale-reading-comprehension","slug":"race-large-scale-reading-comprehension","title":"RACE: Large-scale ReAding Comprehension Dataset From Examinations","date":"2017-04-15","arxiv_id":"1704.04683","repositories_listed":2,"syntology":null},{"url":"/paper/newsqa-a-machine-comprehension-dataset","slug":"newsqa-a-machine-comprehension-dataset","title":"NewsQA: A Machine Comprehension Dataset","date":"2016-11-29","arxiv_id":"1611.09830","repositories_listed":2,"syntology":null},{"url":"/paper/a-compare-aggregate-model-for-matching-text","slug":"a-compare-aggregate-model-for-matching-text","title":"A Compare-Aggregate Model for Matching Text Sequences","date":"2016-11-06","arxiv_id":"1611.01747","repositories_listed":2,"syntology":null},{"url":"/paper/learning-recurrent-span-representations-for","slug":"learning-recurrent-span-representations-for","title":"Learning Recurrent Span Representations for Extractive Question Answering","date":"2016-11-04","arxiv_id":"1611.01436","repositories_listed":2,"syntology":null},{"url":"/paper/embracing-data-abundance-booktest-dataset-for","slug":"embracing-data-abundance-booktest-dataset-for","title":"Embracing data abundance: BookTest Dataset for Reading Comprehension","date":"2016-10-04","arxiv_id":"1610.00956","repositories_listed":2,"syntology":null},{"url":"/paper/do-we-really-need-all-those-rich-linguistic","slug":"do-we-really-need-all-those-rich-linguistic","title":"Do We Really Need All Those Rich Linguistic Features? A Neural Network-Based Approach to Implicit Sense Labeling","date":"2016-08-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/attention-over-attention-neural-networks-for","slug":"attention-over-attention-neural-networks-for","title":"Attention-over-Attention Neural Networks for Reading Comprehension","date":"2016-07-15","arxiv_id":"1607.04423","repositories_listed":2,"syntology":null},{"url":"/paper/text-understanding-with-the-attention-sum","slug":"text-understanding-with-the-attention-sum","title":"Text Understanding with the Attention Sum Reader Network","date":"2016-03-04","arxiv_id":"1603.01547","repositories_listed":2,"syntology":{"n":10,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/text-understanding-with-the-attention-sum#ran","syntology_url":"https://syntology.ai/paper/1603.01547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.01547"}},"official":{"repos":["rkadlec/asreader"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/deris-decoupling-perception-and-cognition-for","slug":"deris-decoupling-perception-and-cognition-for","title":"DeRIS: Decoupling Perception and Cognition for Enhanced Referring Image Segmentation through Loopback Synergy","date":"2025-07-02","arxiv_id":"2507.01738","repositories_listed":1,"syntology":null},{"url":"/paper/chaining-event-spans-for-temporal-relation","slug":"chaining-event-spans-for-temporal-relation","title":"Chaining Event Spans for Temporal Relation Grounding","date":"2025-06-17","arxiv_id":"2506.14213","repositories_listed":1,"syntology":null},{"url":"/paper/comumdr-code-mixed-multi-modal-multi-domain","slug":"comumdr-code-mixed-multi-modal-multi-domain","title":"CoMuMDR: Code-mixed Multi-modal Multi-domain corpus for Discourse paRsing in conversations","date":"2025-06-10","arxiv_id":"2506.08504","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-generation-of-inference-making","slug":"automatic-generation-of-inference-making","title":"Automatic Generation of Inference Making Questions for Reading Comprehension Assessments","date":"2025-06-09","arxiv_id":"2506.08260","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-chunking-and-selection-for-reading","slug":"dynamic-chunking-and-selection-for-reading","title":"Dynamic Chunking and Selection for Reading Comprehension of Ultra-Long Context in Large Language Models","date":"2025-06-01","arxiv_id":"2506.00773","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":1,"n_instrument":8,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-chunking-and-selection-for-reading#ran","syntology_url":"https://syntology.ai/paper/2506.00773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00773"}},"official":{"repos":["ecnu-text-computing/dcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/readbench-measuring-the-dense-text-visual","slug":"readbench-measuring-the-dense-text-visual","title":"ReadBench: Measuring the Dense Text Visual Reading Ability of Vision-Language Models","date":"2025-05-25","arxiv_id":"2505.19091","repositories_listed":1,"syntology":null},{"url":"/paper/self-self-extend-the-context-length-with","slug":"self-self-extend-the-context-length-with","title":"SELF: Self-Extend the Context Length With Logistic Growth Function","date":"2025-05-22","arxiv_id":"2505.17296","repositories_listed":1,"syntology":null},{"url":"/paper/learning-graph-representation-of-agent","slug":"learning-graph-representation-of-agent","title":"Learning Graph Representation of Agent Diffusers","date":"2025-05-10","arxiv_id":"2505.06761","repositories_listed":1,"syntology":null},{"url":"/paper/using-llms-in-generating-design-rationale-for","slug":"using-llms-in-generating-design-rationale-for","title":"Using LLMs in Generating Design Rationale for Software Architecture Decisions","date":"2025-04-29","arxiv_id":"2504.20781","repositories_listed":1,"syntology":null},{"url":"/paper/llm-as-a-judge-reassessing-the-performance-of","slug":"llm-as-a-judge-reassessing-the-performance-of","title":"LLM-as-a-Judge: Reassessing the Performance of LLMs in Extractive QA","date":"2025-04-16","arxiv_id":"2504.11972","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-understand-your-translations","slug":"do-llms-understand-your-translations","title":"Do LLMs Understand Your Translations? Evaluating Paragraph-level MT with Question Answering","date":"2025-04-10","arxiv_id":"2504.07583","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-tuning-of-large-language-models-for","slug":"efficient-tuning-of-large-language-models-for","title":"Efficient Tuning of Large Language Models for Knowledge-Grounded Dialogue Generation","date":"2025-04-10","arxiv_id":"2504.07754","repositories_listed":1,"syntology":null},{"url":"/paper/factguard-leveraging-multi-agent-systems-to","slug":"factguard-leveraging-multi-agent-systems-to","title":"FactGuard: Leveraging Multi-Agent Systems to Generate Answerable and Unanswerable Questions for Enhanced Long-Context LLM Extraction","date":"2025-04-08","arxiv_id":"2504.05607","repositories_listed":1,"syntology":null},{"url":"/paper/locations-of-characters-in-narratives","slug":"locations-of-characters-in-narratives","title":"Locations of Characters in Narratives: Andersen and Persuasion Datasets","date":"2025-04-04","arxiv_id":"2504.03434","repositories_listed":1,"syntology":null},{"url":"/paper/hicd-hallucination-inducing-via-attention","slug":"hicd-hallucination-inducing-via-attention","title":"HICD: Hallucination-Inducing via Attention Dispersion for Contrastive Decoding to Mitigate Hallucinations in Large Language Models","date":"2025-03-17","arxiv_id":"2503.12908","repositories_listed":1,"syntology":null},{"url":"/paper/mrceval-a-comprehensive-challenging-and","slug":"mrceval-a-comprehensive-challenging-and","title":"MRCEval: A Comprehensive, Challenging and Accessible Machine Reading Comprehension Benchmark","date":"2025-03-10","arxiv_id":"2503.07144","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-complex-question-answering-on-long","slug":"zero-shot-complex-question-answering-on-long","title":"Zero-Shot Complex Question-Answering on Long Scientific Documents","date":"2025-03-04","arxiv_id":"2503.02695","repositories_listed":1,"syntology":null},{"url":"/paper/rolemrc-a-fine-grained-composite-benchmark","slug":"rolemrc-a-fine-grained-composite-benchmark","title":"RoleMRC: A Fine-Grained Composite Benchmark for Role-Playing and Instruction-Following","date":"2025-02-17","arxiv_id":"2502.11387","repositories_listed":1,"syntology":null},{"url":"/paper/simlabel-consistency-guided-ood-detection","slug":"simlabel-consistency-guided-ood-detection","title":"SimLabel: Consistency-Guided OOD Detection with Pretrained Vision-Language Models","date":"2025-01-20","arxiv_id":"2501.11485","repositories_listed":1,"syntology":null},{"url":"/paper/biased-or-flawed-mitigating-stereotypes-in","slug":"biased-or-flawed-mitigating-stereotypes-in","title":"Biased or Flawed? Mitigating Stereotypes in Generative Language Models by Addressing Task-Specific Flaws","date":"2024-12-16","arxiv_id":"2412.11414","repositories_listed":1,"syntology":null},{"url":"/paper/asking-again-and-again-exploring-llm","slug":"asking-again-and-again-exploring-llm","title":"Asking Again and Again: Exploring LLM Robustness to Repeated Questions","date":"2024-12-10","arxiv_id":"2412.07923","repositories_listed":1,"syntology":null},{"url":"/paper/scidqa-a-deep-reading-comprehension-dataset","slug":"scidqa-a-deep-reading-comprehension-dataset","title":"SciDQA: A Deep Reading Comprehension Dataset over Scientific Papers","date":"2024-11-08","arxiv_id":"2411.05338","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scidqa-a-deep-reading-comprehension-dataset#ran","syntology_url":"https://syntology.ai/paper/2411.05338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05338"}},"official":{"repos":["yale-nlp/scidqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/diagnosing-medical-datasets-with-training","slug":"diagnosing-medical-datasets-with-training","title":"Diagnosing Medical Datasets with Training Dynamics","date":"2024-11-03","arxiv_id":"2411.01653","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-llms-for-targeted-concept","slug":"evaluating-llms-for-targeted-concept","title":"Evaluating LLMs for Targeted Concept Simplification for Domain-Specific Texts","date":"2024-10-28","arxiv_id":"2410.20763","repositories_listed":1,"syntology":null},{"url":"/paper/llms-are-biased-evaluators-but-not-biased-for","slug":"llms-are-biased-evaluators-but-not-biased-for","title":"LLMs are Biased Evaluators But Not Biased for Retrieval Augmented Generation","date":"2024-10-28","arxiv_id":"2410.20833","repositories_listed":1,"syntology":null},{"url":"/paper/robin-a-transformer-based-model-for-risk-of","slug":"robin-a-transformer-based-model-for-risk-of","title":"RoBIn: A Transformer-Based Model For Risk Of Bias Inference With Machine Reading Comprehension","date":"2024-10-28","arxiv_id":"2410.21495","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-prediction-of-reading","slug":"fine-grained-prediction-of-reading","title":"Fine-Grained Prediction of Reading Comprehension from Eye Movements","date":"2024-10-06","arxiv_id":"2410.04484","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fine-grained-prediction-of-reading#ran","syntology_url":"https://syntology.ai/paper/2410.04484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04484"}},"official":{"repos":["lacclab/Reading-Comprehension-Prediction"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/training-language-models-to-win-debates-with","slug":"training-language-models-to-win-debates-with","title":"Training Language Models to Win Debates with Self-Play Improves Judge Accuracy","date":"2024-09-25","arxiv_id":"2409.16636","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-for-sparse-multidimensional","slug":"data-augmentation-for-sparse-multidimensional","title":"Data Augmentation for Sparse Multidimensional Learning Performance Data Using Generative AI","date":"2024-09-24","arxiv_id":"2409.15631","repositories_listed":1,"syntology":null},{"url":"/paper/thought-path-contrastive-learning-via-premise","slug":"thought-path-contrastive-learning-via-premise","title":"Thought-Path Contrastive Learning via Premise-Oriented Data Augmentation for Logical Reading Comprehension","date":"2024-09-22","arxiv_id":"2409.14495","repositories_listed":1,"syntology":null},{"url":"/paper/shaking-up-vlms-comparing-transformers-and","slug":"shaking-up-vlms-comparing-transformers-and","title":"Shaking Up VLMs: Comparing Transformers and Structured State Space Models for Vision & Language Modeling","date":"2024-09-09","arxiv_id":"2409.05395","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shaking-up-vlms-comparing-transformers-and#ran","syntology_url":"https://syntology.ai/paper/2409.05395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05395"}},"official":{"repos":["gpantaz/vl_mamba"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seemingly-plausible-distractors-in-multi-hop","slug":"seemingly-plausible-distractors-in-multi-hop","title":"Seemingly Plausible Distractors in Multi-Hop Reasoning: Are Large Language Models Attentive Readers?","date":"2024-09-08","arxiv_id":"2409.05197","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/seemingly-plausible-distractors-in-multi-hop#ran","syntology_url":"https://syntology.ai/paper/2409.05197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.05197"}},"official":{"repos":["zawedcvg/are-large-language-models-attentive-readers"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02337","slug":"2408-02337","title":"Developing PUGG for Polish: A Modern Approach to KBQA, MRC, and IR Dataset Construction","date":"2024-08-05","arxiv_id":"2408.02337","repositories_listed":1,"syntology":null},{"url":"/paper/harmonizing-visual-text-comprehension-and","slug":"harmonizing-visual-text-comprehension-and","title":"Harmonizing Visual Text Comprehension and Generation","date":"2024-07-23","arxiv_id":"2407.16364","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/harmonizing-visual-text-comprehension-and#ran","syntology_url":"https://syntology.ai/paper/2407.16364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16364"}},"official":{"repos":["bytedance/textharmony"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-engage-your-readers-generating-guiding","slug":"how-to-engage-your-readers-generating-guiding","title":"How to Engage Your Readers? Generating Guiding Questions to Promote Active Reading","date":"2024-07-19","arxiv_id":"2407.14309","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-to-engage-your-readers-generating-guiding#ran","syntology_url":"https://syntology.ai/paper/2407.14309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14309"}},"official":{"repos":["eth-lre/engage-your-readers"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21033","slug":"2407-21033","title":"Multi-Grained Query-Guided Set Prediction Network for Grounded Multimodal Named Entity Recognition","date":"2024-07-17","arxiv_id":"2407.21033","repositories_listed":1,"syntology":null},{"url":"/paper/docbench-a-benchmark-for-evaluating-llm-based","slug":"docbench-a-benchmark-for-evaluating-llm-based","title":"DOCBENCH: A Benchmark for Evaluating LLM-based Document Reading Systems","date":"2024-07-15","arxiv_id":"2407.10701","repositories_listed":1,"syntology":null},{"url":"/paper/malalgoqa-a-pedagogical-approach-for","slug":"malalgoqa-a-pedagogical-approach-for","title":"MalAlgoQA: Pedagogical Evaluation of Counterfactual Reasoning in Large Language Models and Implications for AI in Education","date":"2024-07-01","arxiv_id":"2407.00938","repositories_listed":1,"syntology":null},{"url":"/paper/fastmem-fast-memorization-of-prompt-improves","slug":"fastmem-fast-memorization-of-prompt-improves","title":"FastMem: Fast Memorization of Prompt Improves Context Awareness of Large Language Models","date":"2024-06-23","arxiv_id":"2406.16069","repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-commonsense-in-language","slug":"improving-visual-commonsense-in-language","title":"Improving Visual Commonsense in Language Models via Multiple Image Generation","date":"2024-06-19","arxiv_id":"2406.13621","repositories_listed":1,"syntology":null},{"url":"/paper/unused-information-in-token-probability","slug":"unused-information-in-token-probability","title":"Unused information in token probability distribution of generative LLM: improving LLM reading comprehension through calculation of expected values","date":"2024-06-11","arxiv_id":"2406.10267","repositories_listed":1,"syntology":null},{"url":"/paper/fairytaleqa-translated-enabling-educational","slug":"fairytaleqa-translated-enabling-educational","title":"FairytaleQA Translated: Enabling Educational Question and Answer Generation in Less-Resourced Languages","date":"2024-06-06","arxiv_id":"2406.04233","repositories_listed":1,"syntology":null},{"url":"/paper/m-qalm-a-benchmark-to-assess-clinical-reading","slug":"m-qalm-a-benchmark-to-assess-clinical-reading","title":"M-QALM: A Benchmark to Assess Clinical Reading Comprehension and Knowledge Recall in Large Language Models via Question Answering","date":"2024-06-06","arxiv_id":"2406.03699","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-temporal-complex-events-with-large","slug":"analyzing-temporal-complex-events-with-large","title":"Analyzing Temporal Complex Events with Large Language Models? A Benchmark towards Temporal, Long Context Understanding","date":"2024-06-04","arxiv_id":"2406.02472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-temporal-complex-events-with-large#ran","syntology_url":"https://syntology.ai/paper/2406.02472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02472"}},"official":{"repos":["Zhihan72/TCELongBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tabpedia-towards-comprehensive-visual-table","slug":"tabpedia-towards-comprehensive-visual-table","title":"TabPedia: Towards Comprehensive Visual Table Understanding with Concept Synergy","date":"2024-06-03","arxiv_id":"2406.01326","repositories_listed":1,"syntology":null},{"url":"/paper/automated-focused-feedback-generation-for","slug":"automated-focused-feedback-generation-for","title":"Automated Focused Feedback Generation for Scientific Writing Assistance","date":"2024-05-30","arxiv_id":"2405.20477","repositories_listed":1,"syntology":null},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-a-multichoice-dataset-be-repurposed-for","slug":"can-a-multichoice-dataset-be-repurposed-for","title":"From Multiple-Choice to Extractive QA: A Case Study for English and Arabic","date":"2024-04-26","arxiv_id":"2404.17342","repositories_listed":1,"syntology":null},{"url":"/paper/noticia-a-clickbait-article-summarization","slug":"noticia-a-clickbait-article-summarization","title":"NoticIA: A Clickbait Article Summarization Dataset in Spanish","date":"2024-04-11","arxiv_id":"2404.07611","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-themes-from-educational-stories","slug":"interpreting-themes-from-educational-stories","title":"Interpreting Themes from Educational Stories","date":"2024-04-08","arxiv_id":"2404.05250","repositories_listed":1,"syntology":null},{"url":"/paper/kazqad-kazakh-open-domain-question-answering","slug":"kazqad-kazakh-open-domain-question-answering","title":"KazQAD: Kazakh Open-Domain Question Answering Dataset","date":"2024-04-06","arxiv_id":"2404.04487","repositories_listed":1,"syntology":null},{"url":"/paper/st-llm-large-language-models-are-effective-1","slug":"st-llm-large-language-models-are-effective-1","title":"ST-LLM: Large Language Models Are Effective Temporal Learners","date":"2024-03-30","arxiv_id":"2404.00308","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/st-llm-large-language-models-are-effective-1#ran","syntology_url":"https://syntology.ai/paper/2404.00308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00308"}},"official":{"repos":["TencentARC/ST-LLM"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/latxa-an-open-language-model-and-evaluation","slug":"latxa-an-open-language-model-and-evaluation","title":"Latxa: An Open Language Model and Evaluation Suite for Basque","date":"2024-03-29","arxiv_id":"2403.20266","repositories_listed":1,"syntology":null},{"url":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic","slug":"arabicaqa-a-comprehensive-dataset-for-arabic","title":"ArabicaQA: A Comprehensive Dataset for Arabic Question Answering","date":"2024-03-26","arxiv_id":"2403.17848","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic#ran","syntology_url":"https://syntology.ai/paper/2403.17848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17848"}},"official":{"repos":["datascienceuibk/arabicaqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chroniclingamericaqa-a-large-scale-question","slug":"chroniclingamericaqa-a-large-scale-question","title":"ChroniclingAmericaQA: A Large-scale Question Answering Dataset based on Historical American Newspaper Pages","date":"2024-03-26","arxiv_id":"2403.17859","repositories_listed":1,"syntology":null},{"url":"/paper/wangchanlion-and-wangchanx-mrc-eval","slug":"wangchanlion-and-wangchanx-mrc-eval","title":"WangchanLion and WangchanX MRC Eval","date":"2024-03-24","arxiv_id":"2403.16127","repositories_listed":1,"syntology":null},{"url":"/paper/ac-eval-evaluating-ancient-chinese-language","slug":"ac-eval-evaluating-ancient-chinese-language","title":"AC-EVAL: Evaluating Ancient Chinese Language Understanding in Large Language Models","date":"2024-03-11","arxiv_id":"2403.06574","repositories_listed":1,"syntology":null},{"url":"/paper/video-relationship-detection-using-mixture-of-1","slug":"video-relationship-detection-using-mixture-of-1","title":"Video Relationship Detection Using Mixture of Experts","date":"2024-03-06","arxiv_id":"2403.03994","repositories_listed":1,"syntology":null},{"url":"/paper/potec-a-german-naturalistic-eye-tracking","slug":"potec-a-german-naturalistic-eye-tracking","title":"PoTeC: A German Naturalistic Eye-tracking-while-reading Corpus","date":"2024-03-01","arxiv_id":"2403.00506","repositories_listed":1,"syntology":null},{"url":"/paper/causal-orthogonalization-multicollinearity","slug":"causal-orthogonalization-multicollinearity","title":"Treatment effects without multicollinearity? Temporal order and the Gram-Schmidt process in causal inference","date":"2024-02-27","arxiv_id":"2402.17103","repositories_listed":1,"syntology":null},{"url":"/paper/vlogqa-task-dataset-and-baseline-models-for","slug":"vlogqa-task-dataset-and-baseline-models-for","title":"VlogQA: Task, Dataset, and Baseline Models for Vietnamese Spoken-Based Machine Reading Comprehension","date":"2024-02-05","arxiv_id":"2402.02655","repositories_listed":1,"syntology":null},{"url":"/paper/an-information-theoretic-approach-to-analyze","slug":"an-information-theoretic-approach-to-analyze","title":"An Information-Theoretic Approach to Analyze NLP Classification Tasks","date":"2024-02-01","arxiv_id":"2402.00978","repositories_listed":1,"syntology":null},{"url":"/paper/do-language-models-exhibit-the-same-cognitive","slug":"do-language-models-exhibit-the-same-cognitive","title":"Do Language Models Exhibit the Same Cognitive Biases in Problem Solving as Human Learners?","date":"2024-01-31","arxiv_id":"2401.18070","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/do-language-models-exhibit-the-same-cognitive#ran","syntology_url":"https://syntology.ai/paper/2401.18070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.18070"}},"official":{"repos":["eth-lre/solving-biases"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-superpositions-of","slug":"large-language-models-are-superpositions-of","title":"Large Language Models are Superpositions of All Characters: Attaining Arbitrary Role-play via Self-Alignment","date":"2024-01-23","arxiv_id":"2401.12474","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-external-knowledge-resources-to","slug":"leveraging-external-knowledge-resources-to","title":"Towards Efficient Methods in Medical Question Answering using Knowledge Graph Embeddings","date":"2024-01-15","arxiv_id":"2401.07977","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-external-knowledge-resources-to#ran","syntology_url":"https://syntology.ai/paper/2401.07977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07977"}},"official":{"repos":["saptarshi059/cdqa-project"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-domain-adaptation-through-extended","slug":"improving-domain-adaptation-through-extended","title":"Improving Domain Adaptation through Extended-Text Reading Comprehension","date":"2024-01-14","arxiv_id":"2401.07284","repositories_listed":1,"syntology":null},{"url":"/paper/avoiding-data-contamination-in-language-model","slug":"avoiding-data-contamination-in-language-model","title":"LatestEval: Addressing Data Contamination in Language Model Evaluation through Dynamic and Time-Sensitive Test Construction","date":"2023-12-19","arxiv_id":"2312.12343","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/avoiding-data-contamination-in-language-model#ran","syntology_url":"https://syntology.ai/paper/2312.12343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12343"}},"official":{"repos":["liyucheng09/latesteval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/do-text-simplification-systems-preserve","slug":"do-text-simplification-systems-preserve","title":"Do Text Simplification Systems Preserve Meaning? A Human Evaluation via Reading Comprehension","date":"2023-12-15","arxiv_id":"2312.10126","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-clinical-reasoners","slug":"large-language-models-are-clinical-reasoners","title":"Large Language Models are Clinical Reasoners: Reasoning-Aware Diagnosis Framework with Prompt-Generated Rationales","date":"2023-12-12","arxiv_id":"2312.07399","repositories_listed":1,"syntology":null}],"record_sha256":"ded94cd843c8cc920bf163656c3ffa3de4ad8c78fb0fc1a2619c5f04b03b250c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}