{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/20","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":109,"rows_per_page":100,"rows":[1901,2000],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/19","next":"/task/question-answering/papers/21","papers":[{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","slug":"artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/artifacts-or-abduction-how-do-llms-answer#ran","syntology_url":"https://syntology.ai/paper/2402.12483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12483"}},"official":{"repos":["nbalepur/mcqa-artifacts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/mars-meaning-aware-response-scoring-for","slug":"mars-meaning-aware-response-scoring-for","title":"MARS: Meaning-Aware Response Scoring for Uncertainty Estimation in Generative LLMs","date":"2024-02-19","arxiv_id":"2402.11756","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mars-meaning-aware-response-scoring-for#ran","syntology_url":"https://syntology.ai/paper/2402.11756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11756"}},"official":{"repos":["ybakman/llm_uncertainity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/small-models-big-insights-leveraging-slim","slug":"small-models-big-insights-leveraging-slim","title":"Small Models, Big Insights: Leveraging Slim Proxy Models To Decide When and What to Retrieve for LLMs","date":"2024-02-19","arxiv_id":"2402.12052","repositories_listed":1,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":15,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/small-models-big-insights-leveraging-slim#ran","syntology_url":"https://syntology.ai/paper/2402.12052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12052"}},"official":{"repos":["plageon/slimplm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/trustscore-reference-free-evaluation-of-llm","slug":"trustscore-reference-free-evaluation-of-llm","title":"TrustScore: Reference-Free Evaluation of LLM Response Trustworthiness","date":"2024-02-19","arxiv_id":"2402.12545","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustscore-reference-free-evaluation-of-llm#ran","syntology_url":"https://syntology.ai/paper/2402.12545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12545"}},"official":{"repos":["dannalily/trustscore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/allava-harnessing-gpt4v-synthesized-data-for","slug":"allava-harnessing-gpt4v-synthesized-data-for","title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","date":"2024-02-18","arxiv_id":"2402.11684","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/allava-harnessing-gpt4v-synthesized-data-for#ran","syntology_url":"https://syntology.ai/paper/2402.11684","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11684"}},"official":{"repos":["freedomintelligence/allava"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/benchmarking-knowledge-boundary-for-large","slug":"benchmarking-knowledge-boundary-for-large","title":"Benchmarking Knowledge Boundary for Large Language Models: A Different Perspective on Model Evaluation","date":"2024-02-18","arxiv_id":"2402.11493","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-knowledge-boundary-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.11493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11493"}},"official":{"repos":["pkulcwmzx/knowledge-boundary"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-failure-integrating-negative","slug":"learning-from-failure-integrating-negative","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","date":"2024-02-18","arxiv_id":"2402.11651","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-failure-integrating-negative#ran","syntology_url":"https://syntology.ai/paper/2402.11651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11651"}},"official":{"repos":["reason-wang/nat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leia-facilitating-cross-lingual-knowledge","slug":"leia-facilitating-cross-lingual-knowledge","title":"LEIA: Facilitating Cross-lingual Knowledge Transfer in Language Models with Entity-based Data Augmentation","date":"2024-02-18","arxiv_id":"2402.11485","repositories_listed":1,"syntology":null},{"url":"/paper/longagent-scaling-language-models-to-128k","slug":"longagent-scaling-language-models-to-128k","title":"LongAgent: Scaling Language Models to 128k Context through Multi-Agent Collaboration","date":"2024-02-18","arxiv_id":"2402.11550","repositories_listed":1,"syntology":null},{"url":"/paper/direct-evaluation-of-chain-of-thought-in","slug":"direct-evaluation-of-chain-of-thought-in","title":"Direct Evaluation of Chain-of-Thought in Multi-hop Reasoning with Knowledge Graphs","date":"2024-02-17","arxiv_id":"2402.11199","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/direct-evaluation-of-chain-of-thought-in#ran","syntology_url":"https://syntology.ai/paper/2402.11199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11199"}},"official":{"repos":["minhvuong2000/llmreasoncert"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/panda-pedantic-answer-correctness","slug":"panda-pedantic-answer-correctness","title":"PEDANTS: Cheap but Effective and Interpretable Answer Equivalence","date":"2024-02-17","arxiv_id":"2402.11161","repositories_listed":1,"syntology":null},{"url":"/paper/ii-mmr-identifying-and-improving-multi-modal","slug":"ii-mmr-identifying-and-improving-multi-modal","title":"II-MMR: Identifying and Improving Multi-modal Multi-hop Reasoning in Visual Question Answering","date":"2024-02-16","arxiv_id":"2402.11058","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-as-science-tutors","slug":"language-models-as-science-tutors","title":"Language Models as Science Tutors","date":"2024-02-16","arxiv_id":"2402.11111","repositories_listed":1,"syntology":null},{"url":"/paper/multi-hop-table-retrieval-for-open-domain","slug":"multi-hop-table-retrieval-for-open-domain","title":"MURRE: Multi-Hop Table Retrieval with Removal for Open-Domain Text-to-SQL","date":"2024-02-16","arxiv_id":"2402.10666","repositories_listed":1,"syntology":null},{"url":"/paper/question-instructed-visual-descriptions-for","slug":"question-instructed-visual-descriptions-for","title":"Question-Instructed Visual Descriptions for Zero-Shot Video Question Answering","date":"2024-02-16","arxiv_id":"2402.10698","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-llm-adaptation-for-question","slug":"unsupervised-llm-adaptation-for-question","title":"Where is the answer? Investigating Positional Bias in Language Model Knowledge Extraction","date":"2024-02-16","arxiv_id":"2402.12170","repositories_listed":1,"syntology":null},{"url":"/paper/ai-hospital-interactive-evaluation-and","slug":"ai-hospital-interactive-evaluation-and","title":"AI Hospital: Benchmarking Large Language Models in a Multi-agent Medical Interaction Simulator","date":"2024-02-15","arxiv_id":"2402.09742","repositories_listed":1,"syntology":null},{"url":"/paper/answer-is-all-you-need-instruction-following","slug":"answer-is-all-you-need-instruction-following","title":"Answer is All You Need: Instruction-following Text Embedding via Answering the Question","date":"2024-02-15","arxiv_id":"2402.09642","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/answer-is-all-you-need-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2402.09642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09642"}},"official":{"repos":["zhang-yu-wei/inbedder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomistral-a-collection-of-open-source","slug":"biomistral-a-collection-of-open-source","title":"BioMistral: A Collection of Open-Source Pretrained Large Language Models for Medical Domains","date":"2024-02-15","arxiv_id":"2402.10373","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/biomistral-a-collection-of-open-source#ran","syntology_url":"https://syntology.ai/paper/2402.10373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10373"}},"official":{"repos":["biomistral/biomistral"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controllm-crafting-diverse-personalities-for","slug":"controllm-crafting-diverse-personalities-for","title":"ControlLM: Crafting Diverse Personalities for Language Models","date":"2024-02-15","arxiv_id":"2402.10151","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/controllm-crafting-diverse-personalities-for#ran","syntology_url":"https://syntology.ai/paper/2402.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10151"}},"official":{"repos":["wengsyx/controllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pretraining-vision-language-model-for","slug":"pretraining-vision-language-model-for","title":"Pretraining Vision-Language Model for Difference Visual Question Answering in Longitudinal Chest X-rays","date":"2024-02-14","arxiv_id":"2402.08966","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-reasoning-in-generative-large","slug":"probabilistic-reasoning-in-generative-large","title":"Reasoning over Uncertain Text by Generative Large Language Models","date":"2024-02-14","arxiv_id":"2402.09614","repositories_listed":1,"syntology":null},{"url":"/paper/plausible-extractive-rationalization-through","slug":"plausible-extractive-rationalization-through","title":"Plausible Extractive Rationalization through Semi-Supervised Entailment Signal","date":"2024-02-13","arxiv_id":"2402.08479","repositories_listed":1,"syntology":null},{"url":"/paper/preflmr-scaling-up-fine-grained-late","slug":"preflmr-scaling-up-fine-grained-late","title":"PreFLMR: Scaling Up Fine-Grained Late-Interaction Multi-modal Retrievers","date":"2024-02-13","arxiv_id":"2402.08327","repositories_listed":1,"syntology":null},{"url":"/paper/towards-faithful-and-robust-llm-specialists","slug":"towards-faithful-and-robust-llm-specialists","title":"Towards Faithful and Robust LLM Specialists for Evidence-Based Question-Answering","date":"2024-02-13","arxiv_id":"2402.08277","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-faithful-and-robust-llm-specialists#ran","syntology_url":"https://syntology.ai/paper/2402.08277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08277"}},"official":{"repos":["EdisonNi-hku/Robust_Evidence_Based_QA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visually-dehallucinative-instruction","slug":"visually-dehallucinative-instruction","title":"Visually Dehallucinative Instruction Generation","date":"2024-02-13","arxiv_id":"2402.08348","repositories_listed":1,"syntology":null},{"url":"/paper/anchor-based-large-language-models","slug":"anchor-based-large-language-models","title":"Anchor-based Large Language Models","date":"2024-02-12","arxiv_id":"2402.07616","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/anchor-based-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.07616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07616"}},"official":{"repos":["pangjh3/anllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-layer-iteratively-prompting-large","slug":"chain-of-layer-iteratively-prompting-large","title":"Chain-of-Layer: Iteratively Prompting Large Language Models for Taxonomy Induction from Limited Examples","date":"2024-02-12","arxiv_id":"2402.07386","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-perceptual-limitation-of-multimodal","slug":"exploring-perceptual-limitation-of-multimodal","title":"Exploring Perceptual Limitation of Multimodal Large Language Models","date":"2024-02-12","arxiv_id":"2402.07384","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-perceptual-limitation-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2402.07384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07384"}},"official":{"repos":["saccharomycetes/mllm-perceptual-limitation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/synthesizing-sentiment-controlled-feedback","slug":"synthesizing-sentiment-controlled-feedback","title":"Synthesizing Sentiment-Controlled Feedback For Multimodal Text and Image Data","date":"2024-02-12","arxiv_id":"2402.07640","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-for-multi-modal-foundation-models","slug":"a-benchmark-for-multi-modal-foundation-models","title":"Q-Bench+: A Benchmark for Multi-modal Foundation Models on Low-level Vision from Single Images to Pairs","date":"2024-02-11","arxiv_id":"2402.07116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-benchmark-for-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2402.07116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07116"}},"official":{"repos":["Q-Future/Q-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graphtranslator-aligning-graph-model-to-large","slug":"graphtranslator-aligning-graph-model-to-large","title":"GraphTranslator: Aligning Graph Model to Large Language Model for Open-ended Tasks","date":"2024-02-11","arxiv_id":"2402.07197","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graphtranslator-aligning-graph-model-to-large#ran","syntology_url":"https://syntology.ai/paper/2402.07197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07197"}},"official":{"repos":["alibaba/graphtranslator"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gemini-goes-to-med-school-exploring-the","slug":"gemini-goes-to-med-school-exploring-the","title":"Gemini Goes to Med School: Exploring the Capabilities of Multimodal Large Language Models on Medical Challenge Problems & Hallucinations","date":"2024-02-10","arxiv_id":"2402.07023","repositories_listed":1,"syntology":null},{"url":"/paper/entgpt-linking-generative-large-language","slug":"entgpt-linking-generative-large-language","title":"EntGPT: Linking Generative Large Language Models with Knowledge Bases","date":"2024-02-09","arxiv_id":"2402.06738","repositories_listed":1,"syntology":null},{"url":"/paper/verif-ai-towards-an-open-source-scientific","slug":"verif-ai-towards-an-open-source-scientific","title":"Verif.ai: Towards an Open-Source Scientific Generative Question-Answering System with Referenced and Verifiable Answers","date":"2024-02-09","arxiv_id":"2402.18589","repositories_listed":1,"syntology":null},{"url":"/paper/crema-multimodal-compositional-video","slug":"crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","arxiv_id":"2402.05889","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crema-multimodal-compositional-video#ran","syntology_url":"https://syntology.ai/paper/2402.05889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05889"}},"official":{"repos":["Yui010206/CREMA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/examining-gender-and-racial-bias-in-large","slug":"examining-gender-and-racial-bias-in-large","title":"Examining Gender and Racial Bias in Large Vision-Language Models Using a Novel Dataset of Parallel Images","date":"2024-02-08","arxiv_id":"2402.05779","repositories_listed":1,"syntology":null},{"url":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","repositories_listed":1,"syntology":null},{"url":"/paper/normy-non-uniform-history-modeling-for-open","slug":"normy-non-uniform-history-modeling-for-open","title":"NORMY: Non-Uniform History Modeling for Open Retrieval Conversational Question Answering","date":"2024-02-07","arxiv_id":"2402.04548","repositories_listed":1,"syntology":null},{"url":"/paper/sparql-generation-an-analysis-on-fine-tuning","slug":"sparql-generation-an-analysis-on-fine-tuning","title":"SPARQL Generation: an analysis on fine-tuning OpenLLaMA for Question Answering over a Life Science Knowledge Graph","date":"2024-02-07","arxiv_id":"2402.04627","repositories_listed":1,"syntology":null},{"url":"/paper/veras-verify-then-assess-stem-lab-reports","slug":"veras-verify-then-assess-stem-lab-reports","title":"VerAs: Verify then Assess STEM Lab Reports","date":"2024-02-07","arxiv_id":"2402.05224","repositories_listed":1,"syntology":null},{"url":"/paper/convincing-rationales-for-visual-question","slug":"convincing-rationales-for-visual-question","title":"Convincing Rationales for Visual Question Answering Reasoning","date":"2024-02-06","arxiv_id":"2402.03896","repositories_listed":1,"syntology":null},{"url":"/paper/inside-llms-internal-states-retain-the-power","slug":"inside-llms-internal-states-retain-the-power","title":"INSIDE: LLMs' Internal States Retain the Power of Hallucination Detection","date":"2024-02-06","arxiv_id":"2402.03744","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/inside-llms-internal-states-retain-the-power#ran","syntology_url":"https://syntology.ai/paper/2402.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03744"}},"official":{"repos":["alibaba/eigenscore"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/training-language-models-to-generate-text","slug":"training-language-models-to-generate-text","title":"Training Language Models to Generate Text with Citations via Fine-grained Rewards","date":"2024-02-06","arxiv_id":"2402.04315","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/training-language-models-to-generate-text#ran","syntology_url":"https://syntology.ai/paper/2402.04315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04315"}},"official":{"repos":["hcy123902/atg-w-fg-rw"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-textbook-question-answering-task","slug":"enhancing-textbook-question-answering-task","title":"Enhancing textual textbook question answering with large language models and retrieval augmented generation","date":"2024-02-05","arxiv_id":"2402.05128","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/enhancing-textbook-question-answering-task#ran","syntology_url":"https://syntology.ai/paper/2402.05128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05128"}},"official":{"repos":["hessaalawwad/plr-tqa"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/text-guided-image-clustering","slug":"text-guided-image-clustering","title":"Text-Guided Image Clustering","date":"2024-02-05","arxiv_id":"2402.02996","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-large-language-models-in-finance","slug":"a-survey-of-large-language-models-in-finance","title":"A Survey of Large Language Models in Finance (FinLLMs)","date":"2024-02-04","arxiv_id":"2402.02315","repositories_listed":1,"syntology":null},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-generation-for-zero-shot-knowledge","slug":"knowledge-generation-for-zero-shot-knowledge","title":"Knowledge Generation for Zero-shot Knowledge-based VQA","date":"2024-02-04","arxiv_id":"2402.02541","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-complex-question-answering-over","slug":"enhancing-complex-question-answering-over","title":"Enhancing Complex Question Answering over Knowledge Graphs through Evidence Pattern Retrieval","date":"2024-02-03","arxiv_id":"2402.02175","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-complex-question-answering-over#ran","syntology_url":"https://syntology.ai/paper/2402.02175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02175"}},"official":{"repos":["nju-websoft/epr-kgqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cabinet-content-relevance-based-noise","slug":"cabinet-content-relevance-based-noise","title":"CABINET: Content Relevance based Noise Reduction for Table Question Answering","date":"2024-02-02","arxiv_id":"2402.01155","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cabinet-content-relevance-based-noise#ran","syntology_url":"https://syntology.ai/paper/2402.01155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01155"}},"official":{"repos":["sohanpatnaik106/cabinet_qa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-makes-a-difference","slug":"instruction-makes-a-difference","title":"Instruction Makes a Difference","date":"2024-02-01","arxiv_id":"2402.00453","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-embodied-interactive-agent-for","slug":"multimodal-embodied-interactive-agent-for","title":"MEIA: Multimodal Embodied Perception and Interaction in Unknown Environments","date":"2024-02-01","arxiv_id":"2402.00290","repositories_listed":1,"syntology":null},{"url":"/paper/sparql-generation-with-entity-pre-trained-gpt","slug":"sparql-generation-with-entity-pre-trained-gpt","title":"SPARQL Generation with Entity Pre-trained GPT for KG Question Answering","date":"2024-02-01","arxiv_id":"2402.00969","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-guided-scene-text-recognition","slug":"instruction-guided-scene-text-recognition","title":"Instruction-Guided Scene Text Recognition","date":"2024-01-31","arxiv_id":"2401.17851","repositories_listed":1,"syntology":null},{"url":"/paper/pipenet-question-answering-with-semantic","slug":"pipenet-question-answering-with-semantic","title":"PipeNet: Question Answering with Semantic Pruning over Knowledge Graphs","date":"2024-01-31","arxiv_id":"2401.17536","repositories_listed":1,"syntology":null},{"url":"/paper/proximity-qa-unleashing-the-power-of-multi","slug":"proximity-qa-unleashing-the-power-of-multi","title":"Proximity QA: Unleashing the Power of Multi-Modal Large Language Models for Spatial Proximity Analysis","date":"2024-01-31","arxiv_id":"2401.17862","repositories_listed":1,"syntology":null},{"url":"/paper/crud-rag-a-comprehensive-chinese-benchmark","slug":"crud-rag-a-comprehensive-chinese-benchmark","title":"CRUD-RAG: A Comprehensive Chinese Benchmark for Retrieval-Augmented Generation of Large Language Models","date":"2024-01-30","arxiv_id":"2401.17043","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crud-rag-a-comprehensive-chinese-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17043"}},"official":{"repos":["iaar-shanghai/crud_rag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qacp-an-annotated-question-answering-dataset","slug":"qacp-an-annotated-question-answering-dataset","title":"QACP: An Annotated Question Answering Dataset for Assisting Chinese Python Programming Learners","date":"2024-01-30","arxiv_id":"2402.07913","repositories_listed":1,"syntology":null},{"url":"/paper/taxonomy-of-mathematical-plagiarism","slug":"taxonomy-of-mathematical-plagiarism","title":"Taxonomy of Mathematical Plagiarism","date":"2024-01-30","arxiv_id":"2401.16969","repositories_listed":1,"syntology":null},{"url":"/paper/ytcommentqa-video-question-answerability-in","slug":"ytcommentqa-video-question-answerability-in","title":"YTCommentQA: Video Question Answerability in Instructional Videos","date":"2024-01-30","arxiv_id":"2401.17343","repositories_listed":1,"syntology":null},{"url":"/paper/augment-before-you-try-knowledge-enhanced","slug":"augment-before-you-try-knowledge-enhanced","title":"Augment before You Try: Knowledge-Enhanced Table Question Answering via Table Expansion","date":"2024-01-28","arxiv_id":"2401.15555","repositories_listed":1,"syntology":null},{"url":"/paper/improving-medical-reasoning-through-retrieval","slug":"improving-medical-reasoning-through-retrieval","title":"Improving Medical Reasoning through Retrieval and Self-Reflection with Retrieval-Augmented Large Language Models","date":"2024-01-27","arxiv_id":"2401.15269","repositories_listed":1,"syntology":null},{"url":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longhealth-a-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.14490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14490"}},"official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-consistent-natural-language","slug":"towards-consistent-natural-language","title":"Towards Consistent Natural-Language Explanations via Explanation-Consistency Finetuning","date":"2024-01-25","arxiv_id":"2401.13986","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-consistent-natural-language#ran","syntology_url":"https://syntology.ai/paper/2401.13986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13986"}},"official":{"repos":["yandachen/explanation-consistency-finetuning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-ai-assistants-know-what-they-don-t-know","slug":"can-ai-assistants-know-what-they-don-t-know","title":"Can AI Assistants Know What They Don't Know?","date":"2024-01-24","arxiv_id":"2401.13275","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-ai-assistants-know-what-they-don-t-know#ran","syntology_url":"https://syntology.ai/paper/2401.13275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13275"}},"official":{"repos":["openmoss/say-i-dont-know"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/clue-guided-path-exploration-an-efficient","slug":"clue-guided-path-exploration-an-efficient","title":"Fine-Grained Stateful Knowledge Exploration: A Novel Paradigm for Integrating Knowledge Graphs with Large Language Models","date":"2024-01-24","arxiv_id":"2401.13444","repositories_listed":1,"syntology":null},{"url":"/paper/instructdoc-a-dataset-for-zero-shot","slug":"instructdoc-a-dataset-for-zero-shot","title":"InstructDoc: A Dataset for Zero-Shot Generalization of Visual Document Understanding with Instructions","date":"2024-01-24","arxiv_id":"2401.13313","repositories_listed":1,"syntology":null},{"url":"/paper/seer-facilitating-structured-reasoning-and","slug":"seer-facilitating-structured-reasoning-and","title":"SEER: Facilitating Structured Reasoning and Explanation via Reinforcement Learning","date":"2024-01-24","arxiv_id":"2401.13246","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seer-facilitating-structured-reasoning-and#ran","syntology_url":"https://syntology.ai/paper/2401.13246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13246"}},"official":{"repos":["chen-gx/seer"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/llmcheckup-conversational-examination-of","slug":"llmcheckup-conversational-examination-of","title":"LLMCheckup: Conversational Examination of Large Language Models via Interpretability Tools and Self-Explanations","date":"2024-01-23","arxiv_id":"2401.12576","repositories_listed":1,"syntology":null},{"url":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trove-inducing-verifiable-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2401.12869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12869"}},"official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/speak-it-out-solving-symbol-related-problems","slug":"speak-it-out-solving-symbol-related-problems","title":"Speak It Out: Solving Symbol-Related Problems with Symbol-to-Language Conversion for Language Models","date":"2024-01-22","arxiv_id":"2401.11725","repositories_listed":1,"syntology":null},{"url":"/paper/medlm-exploring-language-models-for-medical","slug":"medlm-exploring-language-models-for-medical","title":"MedLM: Exploring Language Models for Medical Question Answering Systems","date":"2024-01-21","arxiv_id":"2401.11389","repositories_listed":1,"syntology":null},{"url":"/paper/q-a-prompts-discovering-rich-visual-clues","slug":"q-a-prompts-discovering-rich-visual-clues","title":"Q&A Prompts: Discovering Rich Visual Clues through Mining Question-Answer Prompts for VQA requiring Diverse World Knowledge","date":"2024-01-19","arxiv_id":"2401.10712","repositories_listed":1,"syntology":null},{"url":"/paper/the-radiation-oncology-nlp-database","slug":"the-radiation-oncology-nlp-database","title":"The Radiation Oncology NLP Database","date":"2024-01-19","arxiv_id":"2401.10995","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-gaussian-contrastive","slug":"weakly-supervised-gaussian-contrastive","title":"Weakly Supervised Gaussian Contrastive Grounding with Large Multimodal Models for Video Question Answering","date":"2024-01-19","arxiv_id":"2401.10711","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/weakly-supervised-gaussian-contrastive#ran","syntology_url":"https://syntology.ai/paper/2401.10711","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10711"}},"official":{"repos":["whb139426/gcg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/better-explain-transformers-by-illuminating","slug":"better-explain-transformers-by-illuminating","title":"Better Explain Transformers by Illuminating Important Information","date":"2024-01-18","arxiv_id":"2401.09972","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/better-explain-transformers-by-illuminating#ran","syntology_url":"https://syntology.ai/paper/2401.09972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09972"}},"official":{"repos":["linxins97/mask-lrp"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/veagle-advancements-in-multimodal","slug":"veagle-advancements-in-multimodal","title":"Veagle: Advancements in Multimodal Representation Learning","date":"2024-01-18","arxiv_id":"2403.08773","repositories_listed":1,"syntology":null},{"url":"/paper/mmtom-qa-multimodal-theory-of-mind-question","slug":"mmtom-qa-multimodal-theory-of-mind-question","title":"MMToM-QA: Multimodal Theory of Mind Question Answering","date":"2024-01-16","arxiv_id":"2401.08743","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-large-language-models-limitations","slug":"a-study-on-large-language-models-limitations","title":"A Study on Large Language Models' Limitations in Multiple-Choice Question Answering","date":"2024-01-15","arxiv_id":"2401.07955","repositories_listed":1,"syntology":null},{"url":"/paper/can-large-language-models-explain-themselves-1","slug":"can-large-language-models-explain-themselves-1","title":"Are self-explanations from Large Language Models faithful?","date":"2024-01-15","arxiv_id":"2401.07927","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-explain-themselves-1#ran","syntology_url":"https://syntology.ai/paper/2401.07927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07927"}},"official":{"repos":["AndreasMadsen/llm-introspection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-external-knowledge-resources-to","slug":"leveraging-external-knowledge-resources-to","title":"Towards Efficient Methods in Medical Question Answering using Knowledge Graph Embeddings","date":"2024-01-15","arxiv_id":"2401.07977","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-external-knowledge-resources-to#ran","syntology_url":"https://syntology.ai/paper/2401.07977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07977"}},"official":{"repos":["saptarshi059/cdqa-project"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/survey-of-natural-language-processing-for","slug":"survey-of-natural-language-processing-for","title":"Survey of Natural Language Processing for Education: Taxonomy, Systematic Review, and Future Trends","date":"2024-01-15","arxiv_id":"2401.07518","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ehragent-code-empowers-large-language-models","slug":"ehragent-code-empowers-large-language-models","title":"EHRAgent: Code Empowers Large Language Models for Few-shot Complex Tabular Reasoning on Electronic Health Records","date":"2024-01-13","arxiv_id":"2401.07128","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ehragent-code-empowers-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07128"}},"official":{"repos":["wshi83/ehragent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizing-visual-question-answering-from","slug":"generalizing-visual-question-answering-from","title":"Generalizing Visual Question Answering from Synthetic to Human-Written Questions via a Chain of QA with a Large Language Model","date":"2024-01-12","arxiv_id":"2401.06400","repositories_listed":1,"syntology":null},{"url":"/paper/the-unreasonable-effectiveness-of-easy","slug":"the-unreasonable-effectiveness-of-easy","title":"The Unreasonable Effectiveness of Easy Training Data for Hard Tasks","date":"2024-01-12","arxiv_id":"2401.06751","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-unreasonable-effectiveness-of-easy#ran","syntology_url":"https://syntology.ai/paper/2401.06751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06751"}},"official":{"repos":["allenai/easy-to-hard-generalization"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-large-language-models-via-fine","slug":"improving-large-language-models-via-fine","title":"Improving Large Language Models via Fine-grained Reinforcement Learning with Minimum Editing Constraint","date":"2024-01-11","arxiv_id":"2401.06081","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-large-language-models-via-fine#ran","syntology_url":"https://syntology.ai/paper/2401.06081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06081"}},"official":{"repos":["rucaibox/rlmec"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autoact-automatic-agent-learning-from-scratch","slug":"autoact-automatic-agent-learning-from-scratch","title":"AutoAct: Automatic Agent Learning from Scratch for QA via Self-Planning","date":"2024-01-10","arxiv_id":"2401.05268","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autoact-automatic-agent-learning-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2401.05268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05268"}},"official":{"repos":["zjunlp/autoact"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/miss-a-generative-pretraining-and-finetuning","slug":"miss-a-generative-pretraining-and-finetuning","title":"MISS: A Generative Pretraining and Finetuning Approach for Med-VQA","date":"2024-01-10","arxiv_id":"2401.05163","repositories_listed":1,"syntology":null},{"url":"/paper/answer-retrieval-in-legal-community-question","slug":"answer-retrieval-in-legal-community-question","title":"Answer Retrieval in Legal Community Question Answering","date":"2024-01-09","arxiv_id":"2401.04852","repositories_listed":1,"syntology":null},{"url":"/paper/model-editing-can-hurt-general-abilities-of","slug":"model-editing-can-hurt-general-abilities-of","title":"Model Editing Harms General Abilities of Large Language Models: Regularization to the Rescue","date":"2024-01-09","arxiv_id":"2401.04700","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-editing-can-hurt-general-abilities-of#ran","syntology_url":"https://syntology.ai/paper/2401.04700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04700"}},"official":{"repos":["jasonforjoy/model-editing-hurt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/the-critique-of-critique","slug":"the-critique-of-critique","title":"The Critique of Critique","date":"2024-01-09","arxiv_id":"2401.04518","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-critique-of-critique#ran","syntology_url":"https://syntology.ai/paper/2401.04518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04518"}},"official":{"repos":["gair-nlp/metacritique"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stair-spatial-temporal-reasoning-with","slug":"stair-spatial-temporal-reasoning-with","title":"STAIR: Spatial-Temporal Reasoning with Auditable Intermediate Results for Video Question Answering","date":"2024-01-08","arxiv_id":"2401.03901","repositories_listed":1,"syntology":null},{"url":"/paper/building-efficient-and-effective-openqa","slug":"building-efficient-and-effective-openqa","title":"Building Efficient and Effective OpenQA Systems for Low-Resource Languages","date":"2024-01-07","arxiv_id":"2401.03590","repositories_listed":1,"syntology":null},{"url":"/paper/pefomed-parameter-efficient-fine-tuning-on","slug":"pefomed-parameter-efficient-fine-tuning-on","title":"PeFoMed: Parameter Efficient Fine-tuning of Multimodal Large Language Models for Medical Imaging","date":"2024-01-05","arxiv_id":"2401.02797","repositories_listed":1,"syntology":null}],"record_sha256":"e8025c37836e122d8618b01e377b66005f82f1b03f6e3a776147ef55cc7f1168","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}