{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/24","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":24,"pages_in_order":109,"rows_per_page":100,"rows":[2301,2400],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/23","next":"/task/question-answering/papers/25","papers":[{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/omnidatacomposer-a-unified-data-structure-for","slug":"omnidatacomposer-a-unified-data-structure-for","title":"OmniDataComposer: A Unified Data Structure for Multimodal Data Fusion and Infinite Data Generation","date":"2023-08-08","arxiv_id":"2308.04126","repositories_listed":1,"syntology":null},{"url":"/paper/on-monotonic-aggregation-for-open-domain-qa","slug":"on-monotonic-aggregation-for-open-domain-qa","title":"On Monotonic Aggregation for Open-domain QA","date":"2023-08-08","arxiv_id":"2308.04176","repositories_listed":1,"syntology":null},{"url":"/paper/top-k-relevant-passage-retrieval-for","slug":"top-k-relevant-passage-retrieval-for","title":"Top K Relevant Passage Retrieval for Biomedical Question Answering","date":"2023-08-08","arxiv_id":"2308.04028","repositories_listed":1,"syntology":null},{"url":"/paper/towards-an-ai-to-win-ghana-s-national-science","slug":"towards-an-ai-to-win-ghana-s-national-science","title":"Towards an AI to Win Ghana's National Science and Maths Quiz","date":"2023-08-08","arxiv_id":"2308.04333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-an-ai-to-win-ghana-s-national-science#ran","syntology_url":"https://syntology.ai/paper/2308.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04333"}},"official":{"repos":["nsmq-ai/nsmqai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedical-knowledge-graph-embeddings-with","slug":"biomedical-knowledge-graph-embeddings-with","title":"Biomedical Knowledge Graph Embeddings with Negative Statements","date":"2023-08-07","arxiv_id":"2308.03447","repositories_listed":1,"syntology":null},{"url":"/paper/kitlm-domain-specific-knowledge-integration","slug":"kitlm-domain-specific-knowledge-integration","title":"KITLM: Domain-Specific Knowledge InTegration into Language Models for Question Answering","date":"2023-08-07","arxiv_id":"2308.03638","repositories_listed":1,"syntology":null},{"url":"/paper/no-length-left-behind-enhancing-knowledge","slug":"no-length-left-behind-enhancing-knowledge","title":"No Length Left Behind: Enhancing Knowledge Tracing for Modeling Sequences of Excessive or Insufficient Lengths","date":"2023-08-07","arxiv_id":"2308.03488","repositories_listed":1,"syntology":null},{"url":"/paper/paniniqa-enhancing-patient-education-through","slug":"paniniqa-enhancing-patient-education-through","title":"PaniniQA: Enhancing Patient Education Through Interactive Question Answering","date":"2023-08-07","arxiv_id":"2308.03253","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/paniniqa-enhancing-patient-education-through#ran","syntology_url":"https://syntology.ai/paper/2308.03253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03253"}},"official":{"repos":["pengshancai/paniniqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalist-foundation-model-for","slug":"towards-generalist-foundation-model-for","title":"Towards Generalist Foundation Model for Radiology by Leveraging Web-scale 2D&3D Medical Data","date":"2023-08-04","arxiv_id":"2308.02463","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generalist-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2308.02463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02463"}},"official":{"repos":["chaoyi-wu/radfm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/baby-s-cothought-leveraging-large-language","slug":"baby-s-cothought-leveraging-large-language","title":"Baby's CoThought: Leveraging Large Language Models for Enhanced Reasoning in Compact Models","date":"2023-08-03","arxiv_id":"2308.01684","repositories_listed":1,"syntology":null},{"url":"/paper/conceptlab-creative-generation-using","slug":"conceptlab-creative-generation-using","title":"ConceptLab: Creative Concept Generation using VLM-Guided Diffusion Prior Constraints","date":"2023-08-03","arxiv_id":"2308.02669","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conceptlab-creative-generation-using#ran","syntology_url":"https://syntology.ai/paper/2308.02669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02669"}},"official":{"repos":["kfirgoldberg/ConceptLab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/realcqa-scientific-chart-question-answering","slug":"realcqa-scientific-chart-question-answering","title":"RealCQA: Scientific Chart Question Answering as a Test-bed for First-Order Logic","date":"2023-08-03","arxiv_id":"2308.01979","repositories_listed":1,"syntology":null},{"url":"/paper/the-all-seeing-project-towards-panoptic","slug":"the-all-seeing-project-towards-panoptic","title":"The All-Seeing Project: Towards Panoptic Visual Recognition and Understanding of the Open World","date":"2023-08-03","arxiv_id":"2308.01907","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-smaller-language-models-to","slug":"teaching-smaller-language-models-to","title":"Teaching Smaller Language Models To Generalise To Unseen Compositional Questions","date":"2023-08-02","arxiv_id":"2308.00946","repositories_listed":1,"syntology":null},{"url":"/paper/recomif-reading-comprehension-based-multi","slug":"recomif-reading-comprehension-based-multi","title":"ReCoMIF: Reading comprehension based multi-source information fusion network for Chinese spoken language understanding","date":"2023-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/selfcheck-using-llms-to-zero-shot-check-their","slug":"selfcheck-using-llms-to-zero-shot-check-their","title":"SelfCheck: Using LLMs to Zero-Shot Check Their Own Step-by-Step Reasoning","date":"2023-08-01","arxiv_id":"2308.00436","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/selfcheck-using-llms-to-zero-shot-check-their#ran","syntology_url":"https://syntology.ai/paper/2308.00436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.00436"}},"official":{"repos":["ningmiao/selfcheck"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-correctness-and-faithfulness-of","slug":"evaluating-correctness-and-faithfulness-of","title":"Evaluating Correctness and Faithfulness of Instruction-Following Models for Question Answering","date":"2023-07-31","arxiv_id":"2307.16877","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-correctness-and-faithfulness-of#ran","syntology_url":"https://syntology.ai/paper/2307.16877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16877"}},"official":{"repos":["mcgill-nlp/instruct-qa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/moviechat-from-dense-token-to-sparse-memory","slug":"moviechat-from-dense-token-to-sparse-memory","title":"MovieChat: From Dense Token to Sparse Memory for Long Video Understanding","date":"2023-07-31","arxiv_id":"2307.16449","repositories_listed":1,"syntology":null},{"url":"/paper/no-that-s-not-what-i-meant-handling-third","slug":"no-that-s-not-what-i-meant-handling-third","title":"No that's not what I meant: Handling Third Position Repair in Conversational Question Answering","date":"2023-07-31","arxiv_id":"2307.16689","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-event-centric-knowledge-graphs","slug":"synthesizing-event-centric-knowledge-graphs","title":"Synthesizing Event-centric Knowledge Graphs of Daily Activities Using Virtual Space","date":"2023-07-30","arxiv_id":"2307.16206","repositories_listed":1,"syntology":null},{"url":"/paper/context-vqa-towards-context-aware-and","slug":"context-vqa-towards-context-aware-and","title":"Context-VQA: Towards Context-Aware and Purposeful Visual Question Answering","date":"2023-07-28","arxiv_id":"2307.15745","repositories_listed":1,"syntology":null},{"url":"/paper/rt-2-vision-language-action-models-transfer","slug":"rt-2-vision-language-action-models-transfer","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","date":"2023-07-28","arxiv_id":"2307.15818","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-spatio-temporal-rationales-for","slug":"discovering-spatio-temporal-rationales-for","title":"Discovering Spatio-Temporal Rationales for Video Question Answering","date":"2023-07-22","arxiv_id":"2307.12058","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/discovering-spatio-temporal-rationales-for#ran","syntology_url":"https://syntology.ai/paper/2307.12058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12058"}},"official":{"repos":["yl3800/transtr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generator-retriever-generator-a-novel","slug":"generator-retriever-generator-a-novel","title":"Generator-Retriever-Generator Approach for Open-Domain Question Answering","date":"2023-07-21","arxiv_id":"2307.11278","repositories_listed":1,"syntology":null},{"url":"/paper/mythqa-query-based-large-scale-check-worthy","slug":"mythqa-query-based-large-scale-check-worthy","title":"MythQA: Query-Based Large-Scale Check-Worthy Claim Detection through Multi-Answer Open-Domain Question Answering","date":"2023-07-21","arxiv_id":"2307.11848","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-factual-knowledge-boundary","slug":"investigating-the-factual-knowledge-boundary","title":"Investigating the Factual Knowledge Boundary of Large Language Models with Retrieval Augmentation","date":"2023-07-20","arxiv_id":"2307.11019","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-the-factual-knowledge-boundary#ran","syntology_url":"https://syntology.ai/paper/2307.11019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11019"}},"official":{"repos":["rucaibox/llm-knowledge-boundary"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-autonomous-driving-actions-with","slug":"explaining-autonomous-driving-actions-with","title":"Explaining Autonomous Driving Actions with Visual Question Answering","date":"2023-07-19","arxiv_id":"2307.10408","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-performance-analysis-on-pre-trained","slug":"towards-a-performance-analysis-on-pre-trained","title":"Towards a performance analysis on pre-trained Visual Question Answering models for autonomous driving","date":"2023-07-18","arxiv_id":"2307.09329","repositories_listed":1,"syntology":null},{"url":"/paper/question-decomposition-improves-the","slug":"question-decomposition-improves-the","title":"Question Decomposition Improves the Faithfulness of Model-Generated Reasoning","date":"2023-07-17","arxiv_id":"2307.11768","repositories_listed":1,"syntology":null},{"url":"/paper/coupling-large-language-models-with-logic","slug":"coupling-large-language-models-with-logic","title":"Coupling Large Language Models with Logic Programming for Robust and General Reasoning from Text","date":"2023-07-15","arxiv_id":"2307.07696","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coupling-large-language-models-with-logic#ran","syntology_url":"https://syntology.ai/paper/2307.07696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07696"}},"official":{"repos":["azreasoners/llm-asp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decompeval-evaluating-generated-texts-as","slug":"decompeval-evaluating-generated-texts-as","title":"DecompEval: Evaluating Generated Texts as Unsupervised Decomposed Question Answering","date":"2023-07-13","arxiv_id":"2307.06869","repositories_listed":1,"syntology":null},{"url":"/paper/polylm-an-open-source-polyglot-large-language","slug":"polylm-an-open-source-polyglot-large-language","title":"PolyLM: An Open Source Polyglot Large Language Model","date":"2023-07-12","arxiv_id":"2307.06018","repositories_listed":1,"syntology":null},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egovlpv2-egocentric-video-language-pre","slug":"egovlpv2-egocentric-video-language-pre","title":"EgoVLPv2: Egocentric Video-Language Pre-training with Fusion in the Backbone","date":"2023-07-11","arxiv_id":"2307.05463","repositories_listed":1,"syntology":null},{"url":"/paper/one-versus-others-attention-scalable","slug":"one-versus-others-attention-scalable","title":"One-Versus-Others Attention: Scalable Multimodal Integration for Biomedical Data","date":"2023-07-11","arxiv_id":"2307.05435","repositories_listed":1,"syntology":null},{"url":"/paper/rad-restruct-a-novel-vqa-benchmark-and-method","slug":"rad-restruct-a-novel-vqa-benchmark-and-method","title":"Rad-ReStruct: A Novel VQA Benchmark and Method for Structured Radiology Reporting","date":"2023-07-11","arxiv_id":"2307.05766","repositories_listed":1,"syntology":null},{"url":"/paper/beavertails-towards-improved-safety-alignment-1","slug":"beavertails-towards-improved-safety-alignment-1","title":"BeaverTails: Towards Improved Safety Alignment of LLM via a Human-Preference Dataset","date":"2023-07-10","arxiv_id":"2307.04657","repositories_listed":1,"syntology":null},{"url":"/paper/multi-granularity-temporal-question-answering","slug":"multi-granularity-temporal-question-answering","title":"Multi-granularity Temporal Question Answering over Knowledge Graphs","date":"2023-07-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/answering-ambiguous-questions-via-iterative","slug":"answering-ambiguous-questions-via-iterative","title":"Answering Ambiguous Questions via Iterative Prompting","date":"2023-07-08","arxiv_id":"2307.03897","repositories_listed":1,"syntology":null},{"url":"/paper/reading-between-the-lanes-text-videoqa-on-the","slug":"reading-between-the-lanes-text-videoqa-on-the","title":"Reading Between the Lanes: Text VideoQA on the Road","date":"2023-07-08","arxiv_id":"2307.03948","repositories_listed":1,"syntology":null},{"url":"/paper/trac-trustworthy-retrieval-augmented-chatbot","slug":"trac-trustworthy-retrieval-augmented-chatbot","title":"TRAQ: Trustworthy Retrieval Augmented Question Answering via Conformal Prediction","date":"2023-07-07","arxiv_id":"2307.04642","repositories_listed":1,"syntology":null},{"url":"/paper/core-gpt-combining-open-access-research-and","slug":"core-gpt-combining-open-access-research-and","title":"CORE-GPT: Combining Open Access research and large language models for credible, trustworthy question answering","date":"2023-07-06","arxiv_id":"2307.04683","repositories_listed":1,"syntology":null},{"url":"/paper/improving-retrieval-augmented-large-language","slug":"improving-retrieval-augmented-large-language","title":"Improving Retrieval-Augmented Large Language Models via Data Importance Learning","date":"2023-07-06","arxiv_id":"2307.03027","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-retrieval-augmented-large-language#ran","syntology_url":"https://syntology.ai/paper/2307.03027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03027"}},"official":{"repos":["amsterdata/ragbooster"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/prd-peer-rank-and-discussion-improve-large","slug":"prd-peer-rank-and-discussion-improve-large","title":"PRD: Peer Rank and Discussion Improve Large Language Model based Evaluations","date":"2023-07-06","arxiv_id":"2307.02762","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prd-peer-rank-and-discussion-improve-large#ran","syntology_url":"https://syntology.ai/paper/2307.02762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02762"}},"official":{"repos":["bcdnlp/prd"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/recallm-an-architecture-for-temporal-context","slug":"recallm-an-architecture-for-temporal-context","title":"RecallM: An Adaptable Memory Mechanism with Temporal Understanding for Large Language Models","date":"2023-07-06","arxiv_id":"2307.02738","repositories_listed":1,"syntology":null},{"url":"/paper/text-alignment-is-an-efficient-unified-model","slug":"text-alignment-is-an-efficient-unified-model","title":"Text Alignment Is An Efficient Unified Model for Massive NLP Tasks","date":"2023-07-06","arxiv_id":"2307.02729","repositories_listed":1,"syntology":null},{"url":"/paper/won-t-get-fooled-again-answering-questions","slug":"won-t-get-fooled-again-answering-questions","title":"Won't Get Fooled Again: Answering Questions with False Premises","date":"2023-07-05","arxiv_id":"2307.02394","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/won-t-get-fooled-again-answering-questions#ran","syntology_url":"https://syntology.ai/paper/2307.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02394"}},"official":{"repos":["thunlp/falseqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/journeydb-a-benchmark-for-generative-image","slug":"journeydb-a-benchmark-for-generative-image","title":"JourneyDB: A Benchmark for Generative Image Understanding","date":"2023-07-03","arxiv_id":"2307.00716","repositories_listed":1,"syntology":null},{"url":"/paper/localized-questions-in-medical-visual","slug":"localized-questions-in-medical-visual","title":"Localized Questions in Medical Visual Question Answering","date":"2023-07-03","arxiv_id":"2307.01067","repositories_listed":1,"syntology":null},{"url":"/paper/make-text-unlearnable-exploiting-effective","slug":"make-text-unlearnable-exploiting-effective","title":"Make Text Unlearnable: Exploiting Effective Patterns to Protect Personal Data","date":"2023-07-02","arxiv_id":"2307.00456","repositories_listed":1,"syntology":null},{"url":"/paper/batgpt-a-bidirectional-autoregessive-talker","slug":"batgpt-a-bidirectional-autoregessive-talker","title":"BatGPT: A Bidirectional Autoregessive Talker from Generative Pre-trained Transformer","date":"2023-07-01","arxiv_id":"2307.00360","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-recurrent-cross-modal-encoder-for","slug":"lightweight-recurrent-cross-modal-encoder-for","title":"Lightweight Recurrent Cross-modal Encoder for Video Question Answering","date":"2023-06-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-prompt-retrieval-for-generative","slug":"multimodal-prompt-retrieval-for-generative","title":"Multimodal Prompt Retrieval for Generative Visual Question Answering","date":"2023-06-30","arxiv_id":"2306.17675","repositories_listed":1,"syntology":null},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-multi-modal-dense-retrievers-for","slug":"pre-training-multi-modal-dense-retrievers-for","title":"Pre-Training Multi-Modal Dense Retrievers for Outside-Knowledge Visual Question Answering","date":"2023-06-28","arxiv_id":"2306.16478","repositories_listed":1,"syntology":null},{"url":"/paper/se-pqa-personalized-community-question","slug":"se-pqa-personalized-community-question","title":"SE-PQA: Personalized Community Question Answering","date":"2023-06-28","arxiv_id":"2306.16261","repositories_listed":1,"syntology":null},{"url":"/paper/fauno-the-italian-large-language-model-that","slug":"fauno-the-italian-large-language-model-that","title":"Fauno: The Italian Large Language Model that will leave you senza parole!","date":"2023-06-26","arxiv_id":"2306.14457","repositories_listed":1,"syntology":null},{"url":"/paper/funqa-towards-surprising-video-comprehension","slug":"funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","arxiv_id":"2306.14899","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funqa-towards-surprising-video-comprehension#ran","syntology_url":"https://syntology.ai/paper/2306.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14899"}},"official":{"repos":["jingkang50/funqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robut-a-systematic-study-of-table-qa","slug":"robut-a-systematic-study-of-table-qa","title":"RobuT: A Systematic Study of Table QA Robustness Against Human-Annotated Adversarial Perturbations","date":"2023-06-25","arxiv_id":"2306.14321","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robut-a-systematic-study-of-table-qa#ran","syntology_url":"https://syntology.ai/paper/2306.14321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14321"}},"official":{"repos":["yilunzhao/robut"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tracking-public-attitudes-toward-chatgpt-on","slug":"tracking-public-attitudes-toward-chatgpt-on","title":"Public Attitudes Toward ChatGPT on Twitter: Sentiments, Topics, and Occupations","date":"2023-06-22","arxiv_id":"2306.12951","repositories_listed":1,"syntology":null},{"url":"/paper/ecg-qa-a-comprehensive-question-answering","slug":"ecg-qa-a-comprehensive-question-answering","title":"ECG-QA: A Comprehensive Question Answering Dataset Combined With Electrocardiogram","date":"2023-06-21","arxiv_id":"2306.15681","repositories_listed":1,"syntology":null},{"url":"/paper/resources-and-evaluations-for-multi","slug":"resources-and-evaluations-for-multi","title":"Resources and Evaluations for Multi-Distribution Dense Information Retrieval","date":"2023-06-21","arxiv_id":"2306.12601","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cosa-concatenated-sample-pretrained-vision","slug":"cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","arxiv_id":"2306.09085","repositories_listed":1,"syntology":null},{"url":"/paper/encyclopedic-vqa-visual-questions-about","slug":"encyclopedic-vqa-visual-questions-about","title":"Encyclopedic VQA: Visual questions about detailed properties of fine-grained categories","date":"2023-06-15","arxiv_id":"2306.09224","repositories_listed":1,"syntology":null},{"url":"/paper/lvlm-ehub-a-comprehensive-evaluation","slug":"lvlm-ehub-a-comprehensive-evaluation","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","date":"2023-06-15","arxiv_id":"2306.09265","repositories_listed":1,"syntology":null},{"url":"/paper/neural-models-for-factual-inconsistency","slug":"neural-models-for-factual-inconsistency","title":"Neural models for Factual Inconsistency Classification with Explanations","date":"2023-06-15","arxiv_id":"2306.08872","repositories_listed":1,"syntology":null},{"url":"/paper/towards-benchmarking-and-improving-the","slug":"towards-benchmarking-and-improving-the","title":"Towards Benchmarking and Improving the Temporal Reasoning Capability of Large Language Models","date":"2023-06-15","arxiv_id":"2306.08952","repositories_listed":1,"syntology":null},{"url":"/paper/improving-selective-visual-question-answering-1","slug":"improving-selective-visual-question-answering-1","title":"Improving Selective Visual Question Answering by Learning from Your Peers","date":"2023-06-14","arxiv_id":"2306.08751","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/question-decomposition-tree-for-answering","slug":"question-decomposition-tree-for-answering","title":"Question Decomposition Tree for Answering Complex Questions over Knowledge Bases","date":"2023-06-13","arxiv_id":"2306.07597","repositories_listed":1,"syntology":null},{"url":"/paper/safeguarding-data-in-multimodal-ai-a","slug":"safeguarding-data-in-multimodal-ai-a","title":"Safeguarding Data in Multimodal AI: A Differentially Private Approach to CLIP Training","date":"2023-06-13","arxiv_id":"2306.08173","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-semantic-completion-learning","slug":"global-and-local-semantic-completion-learning","title":"Global and Local Semantic Completion Learning for Vision-Language Pre-training","date":"2023-06-12","arxiv_id":"2306.07096","repositories_listed":1,"syntology":null},{"url":"/paper/the-effect-of-masking-strategies-on-knowledge","slug":"the-effect-of-masking-strategies-on-knowledge","title":"The Effect of Masking Strategies on Knowledge Retention by Language Models","date":"2023-06-12","arxiv_id":"2306.07185","repositories_listed":1,"syntology":null},{"url":"/paper/when-do-annotator-demographics-matter","slug":"when-do-annotator-demographics-matter","title":"When Do Annotator Demographics Matter? Measuring the Influence of Annotator Demographics with the POPQUORN Dataset","date":"2023-06-12","arxiv_id":"2306.06826","repositories_listed":1,"syntology":null},{"url":"/paper/multi-source-test-time-adaptation-as-dueling","slug":"multi-source-test-time-adaptation-as-dueling","title":"Multi-Source Test-Time Adaptation as Dueling Bandits for Extractive Question Answering","date":"2023-06-11","arxiv_id":"2306.06779","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-pre-training-for-medical-vision","slug":"multi-modal-pre-training-for-medical-vision","title":"Multi-modal Pre-training for Medical Vision-language Understanding and Generation: An Empirical Study with A New Benchmark","date":"2023-06-10","arxiv_id":"2306.06494","repositories_listed":1,"syntology":null},{"url":"/paper/modular-visual-question-answering-via-code","slug":"modular-visual-question-answering-via-code","title":"Modular Visual Question Answering via Code Generation","date":"2023-06-08","arxiv_id":"2306.05392","repositories_listed":1,"syntology":null},{"url":"/paper/gotta-generative-few-shot-question-answering","slug":"gotta-generative-few-shot-question-answering","title":"Gotta: Generative Few-shot Question Answering by Prompt-based Cloze Data Augmentation","date":"2023-06-07","arxiv_id":"2306.04101","repositories_listed":1,"syntology":null},{"url":"/paper/phrase-retrieval-for-open-domain","slug":"phrase-retrieval-for-open-domain","title":"Phrase Retrieval for Open-Domain Conversational Question Answering with Conversational Dependency Modeling via Contrastive Learning","date":"2023-06-07","arxiv_id":"2306.04293","repositories_listed":1,"syntology":null},{"url":"/paper/an-approach-to-solving-the-abstraction-and","slug":"an-approach-to-solving-the-abstraction-and","title":"An Approach to Solving the Abstraction and Reasoning Corpus (ARC) Challenge","date":"2023-06-06","arxiv_id":"2306.03553","repositories_listed":1,"syntology":null},{"url":"/paper/cue-an-uncertainty-interpretation-framework","slug":"cue-an-uncertainty-interpretation-framework","title":"CUE: An Uncertainty Interpretation Framework for Text Classifiers Built on Pre-Trained Language Models","date":"2023-06-06","arxiv_id":"2306.03598","repositories_listed":1,"syntology":null},{"url":"/paper/logiqa-2-0-an-improved-dataset-for-logical","slug":"logiqa-2-0-an-improved-dataset-for-logical","title":"LogiQA 2.0—An Improved Dataset for Logical Reasoning in Natural Language Understanding","date":"2023-06-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/prompt-space-optimizing-few-shot-reasoning","slug":"prompt-space-optimizing-few-shot-reasoning","title":"Prompt Space Optimizing Few-shot Reasoning Success with Large Language Models","date":"2023-06-06","arxiv_id":"2306.03799","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-space-optimizing-few-shot-reasoning#ran","syntology_url":"https://syntology.ai/paper/2306.03799","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03799"}},"official":{"repos":["youblei/prompt-space"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-how-to-specialize-large-vision-language-1","slug":"q-how-to-specialize-large-vision-language-1","title":"Q: How to Specialize Large Vision-Language Models to Data-Scarce VQA Tasks? A: Self-Train on Unlabeled Images!","date":"2023-06-06","arxiv_id":"2306.03932","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-good-towards-distribution-shift-evaluation","slug":"do-good-towards-distribution-shift-evaluation","title":"Do-GOOD: Towards Distribution Shift Evaluation for Pre-Trained Visual Document Understanding Models","date":"2023-06-05","arxiv_id":"2306.02623","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-human-feedback-gives-better","slug":"fine-grained-human-feedback-gives-better","title":"Fine-Grained Human Feedback Gives Better Rewards for Language Model Training","date":"2023-06-02","arxiv_id":"2306.01693","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fine-grained-human-feedback-gives-better#ran","syntology_url":"https://syntology.ai/paper/2306.01693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01693"}},"official":null}},{"url":"/paper/visualgptscore-visio-linguistic-reasoning","slug":"visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01879","repositories_listed":1,"syntology":null},{"url":"/paper/llava-med-training-a-large-language-and","slug":"llava-med-training-a-large-language-and","title":"LLaVA-Med: Training a Large Language-and-Vision Assistant for Biomedicine in One Day","date":"2023-06-01","arxiv_id":"2306.00890","repositories_listed":1,"syntology":null},{"url":"/paper/make-pre-trained-model-reversible-from-1","slug":"make-pre-trained-model-reversible-from-1","title":"Make Pre-trained Model Reversible: From Parameter to Memory Efficient Fine-Tuning","date":"2023-06-01","arxiv_id":"2306.00477","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/make-pre-trained-model-reversible-from-1#ran","syntology_url":"https://syntology.ai/paper/2306.00477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00477"}},"official":{"repos":["baohaoliao/mefts"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/timelineqa-a-benchmark-for-question-answering","slug":"timelineqa-a-benchmark-for-question-answering","title":"TimelineQA: A Benchmark for Question Answering over Timelines","date":"2023-06-01","arxiv_id":"2306.01069","repositories_listed":1,"syntology":null},{"url":"/paper/dense-and-aligned-captions-dac-promote","slug":"dense-and-aligned-captions-dac-promote","title":"Dense and Aligned Captions (DAC) Promote Compositional Reasoning in VL Models","date":"2023-05-31","arxiv_id":"2305.19595","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-base-question-answering-for-space","slug":"knowledge-base-question-answering-for-space","title":"Knowledge Base Question Answering for Space Debris Queries","date":"2023-05-31","arxiv_id":"2305.19734","repositories_listed":1,"syntology":null},{"url":"/paper/ukp-square-an-interactive-tool-for-teaching","slug":"ukp-square-an-interactive-tool-for-teaching","title":"UKP-SQuARE: An Interactive Tool for Teaching Question Answering","date":"2023-05-31","arxiv_id":"2305.19748","repositories_listed":1,"syntology":null},{"url":"/paper/a-template-independent-approach-for","slug":"a-template-independent-approach-for","title":"A template-independent approach for information extraction in real estate documents","date":"2023-05-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/concise-answers-to-complex-questions","slug":"concise-answers-to-complex-questions","title":"Concise Answers to Complex Questions: Summarization of Long-form Answers","date":"2023-05-30","arxiv_id":"2305.19271","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/concise-answers-to-complex-questions#ran","syntology_url":"https://syntology.ai/paper/2305.19271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19271"}},"official":{"repos":["acpotluri/lfqa_summary"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"86a855126d5bc66f3adaea6877d210e9b96f49b746e81dbba0284a092fc108c6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}