{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/19","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":109,"rows_per_page":100,"rows":[1801,1900],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/18","next":"/task/question-answering/papers/20","papers":[{"url":"/paper/procqa-a-large-scale-community-based","slug":"procqa-a-large-scale-community-based","title":"ProCQA: A Large-scale Community-based Programming Question Answering Dataset for Code Search","date":"2024-03-25","arxiv_id":"2403.16702","repositories_listed":1,"syntology":null},{"url":"/paper/visually-guided-generative-text-layout-pre","slug":"visually-guided-generative-text-layout-pre","title":"Visually Guided Generative Text-Layout Pre-training for Document Intelligence","date":"2024-03-25","arxiv_id":"2403.16516","repositories_listed":1,"syntology":null},{"url":"/paper/blended-rag-improving-rag-retriever-augmented","slug":"blended-rag-improving-rag-retriever-augmented","title":"Blended RAG: Improving RAG (Retriever-Augmented Generation) Accuracy with Semantic Search and Hybrid Query-Based Retrievers","date":"2024-03-22","arxiv_id":"2404.07220","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blended-rag-improving-rag-retriever-augmented#ran","syntology_url":"https://syntology.ai/paper/2404.07220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07220"}},"official":{"repos":["ibm-ecosystem-engineering/blended-rag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imagination-augmented-generation-learning-to","slug":"imagination-augmented-generation-learning-to","title":"Awakening Augmented Generation: Learning to Awaken Internal Knowledge of Large Language Models for Question Answering","date":"2024-03-22","arxiv_id":"2403.15268","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/imagination-augmented-generation-learning-to#ran","syntology_url":"https://syntology.ai/paper/2403.15268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15268"}},"official":{"repos":["xnhyacinth/iag"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-prumerge-adaptive-token-reduction-for#ran","syntology_url":"https://syntology.ai/paper/2403.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15388"}},"official":null}},{"url":"/paper/language-repository-for-long-video","slug":"language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","arxiv_id":"2403.14622","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-repository-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2403.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14622"}},"official":{"repos":["kkahatapitiya/langrepo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-vqa-exploring-multi-agent","slug":"multi-agent-vqa-exploring-multi-agent","title":"Multi-Agent VQA: Exploring Multi-Agent Foundation Models in Zero-Shot Visual Question Answering","date":"2024-03-21","arxiv_id":"2403.14783","repositories_listed":1,"syntology":null},{"url":"/paper/desire-me-domain-enhanced-supervised","slug":"desire-me-domain-enhanced-supervised","title":"DESIRE-ME: Domain-Enhanced Supervised Information REtrieval using Mixture-of-Experts","date":"2024-03-20","arxiv_id":"2403.13468","repositories_listed":1,"syntology":null},{"url":"/paper/alphafin-benchmarking-financial-analysis-with","slug":"alphafin-benchmarking-financial-analysis-with","title":"AlphaFin: Benchmarking Financial Analysis with Retrieval-Augmented Stock-Chain Framework","date":"2024-03-19","arxiv_id":"2403.12582","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alphafin-benchmarking-financial-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2403.12582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12582"}},"official":{"repos":["alphafin-proj/alphafin"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dr3-ask-large-language-models-not-to-give-off","slug":"dr3-ask-large-language-models-not-to-give-off","title":"Dr3: Ask Large Language Models Not to Give Off-Topic Answers in Open Domain Multi-Hop Question Answering","date":"2024-03-19","arxiv_id":"2403.12393","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dr3-ask-large-language-models-not-to-give-off#ran","syntology_url":"https://syntology.ai/paper/2403.12393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12393"}},"official":{"repos":["gy915/dr3"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/encode-once-and-decode-in-parallel-efficient","slug":"encode-once-and-decode-in-parallel-efficient","title":"Efficient Encoder-Decoder Transformer Decoding for Decomposable Tasks","date":"2024-03-19","arxiv_id":"2403.13112","repositories_listed":1,"syntology":null},{"url":"/paper/seven-pruning-transformer-model-by-reserving","slug":"seven-pruning-transformer-model-by-reserving","title":"SEVEN: Pruning Transformer Model by Reserving Sentinels","date":"2024-03-19","arxiv_id":"2403.12688","repositories_listed":1,"syntology":null},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/novelqa-a-benchmark-for-long-range-novel","slug":"novelqa-a-benchmark-for-long-range-novel","title":"NovelQA: Benchmarking Question Answering on Documents Exceeding 200K Tokens","date":"2024-03-18","arxiv_id":"2403.12766","repositories_listed":1,"syntology":null},{"url":"/paper/syn-qa2-evaluating-false-assumptions-in-long","slug":"syn-qa2-evaluating-false-assumptions-in-long","title":"Syn-QA2: Evaluating False Assumptions in Long-tail Questions with Synthetic QA Datasets","date":"2024-03-18","arxiv_id":"2403.12145","repositories_listed":1,"syntology":null},{"url":"/paper/logic-query-of-thoughts-guiding-large","slug":"logic-query-of-thoughts-guiding-large","title":"Logic Query of Thoughts: Guiding Large Language Models to Answer Complex Logic Queries with Knowledge Graphs","date":"2024-03-17","arxiv_id":"2404.04264","repositories_listed":1,"syntology":null},{"url":"/paper/benqa-a-question-answering-and-reasoning","slug":"benqa-a-question-answering-and-reasoning","title":"BEnQA: A Question Answering and Reasoning Benchmark for Bengali and English","date":"2024-03-16","arxiv_id":"2403.10900","repositories_listed":1,"syntology":null},{"url":"/paper/forward-learning-of-graph-neural-networks","slug":"forward-learning-of-graph-neural-networks","title":"Forward Learning of Graph Neural Networks","date":"2024-03-16","arxiv_id":"2403.11004","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forward-learning-of-graph-neural-networks#ran","syntology_url":"https://syntology.ai/paper/2403.11004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11004"}},"official":{"repos":["facebookresearch/forwardgnn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retinaqa-a-knowledge-base-question-answering","slug":"retinaqa-a-knowledge-base-question-answering","title":"RetinaQA: A Robust Knowledge Base Question Answering Model for both Answerable and Unanswerable Questions","date":"2024-03-16","arxiv_id":"2403.10849","repositories_listed":1,"syntology":null},{"url":"/paper/team-trifecta-at-factify5wqa-setting-the","slug":"team-trifecta-at-factify5wqa-setting-the","title":"Team Trifecta at Factify5WQA: Setting the Standard in Fact Verification with Fine-Tuning","date":"2024-03-15","arxiv_id":"2403.10281","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-training-with-ocr-modality","slug":"adversarial-training-with-ocr-modality","title":"Adversarial Training with OCR Modality Perturbation for Scene-Text Visual Question Answering","date":"2024-03-14","arxiv_id":"2403.09288","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-language-models-texture-or-shape","slug":"are-vision-language-models-texture-or-shape","title":"Can We Talk Models Into Seeing the World Differently?","date":"2024-03-14","arxiv_id":"2403.09193","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/are-vision-language-models-texture-or-shape#ran","syntology_url":"https://syntology.ai/paper/2403.09193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09193"}},"official":{"repos":["paulgavrikov/vlm_shapebias"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/chartinstruct-instruction-tuning-for-chart","slug":"chartinstruct-instruction-tuning-for-chart","title":"ChartInstruct: Instruction Tuning for Chart Comprehension and Reasoning","date":"2024-03-14","arxiv_id":"2403.09028","repositories_listed":1,"syntology":null},{"url":"/paper/ragged-towards-informed-design-of-retrieval","slug":"ragged-towards-informed-design-of-retrieval","title":"RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems","date":"2024-03-14","arxiv_id":"2403.09040","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ragged-towards-informed-design-of-retrieval#ran","syntology_url":"https://syntology.ai/paper/2403.09040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09040"}},"official":{"repos":["neulab/ragged"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-augmented-text-to-sql-generation","slug":"retrieval-augmented-text-to-sql-generation","title":"Retrieval augmented text-to-SQL generation for epidemiological question answering using electronic health records","date":"2024-03-14","arxiv_id":"2403.09226","repositories_listed":1,"syntology":null},{"url":"/paper/transformers-get-stable-an-end-to-end-signal","slug":"transformers-get-stable-an-end-to-end-signal","title":"Transformers Get Stable: An End-to-End Signal Propagation Theory for Language Models","date":"2024-03-14","arxiv_id":"2403.09635","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transformers-get-stable-an-end-to-end-signal#ran","syntology_url":"https://syntology.ai/paper/2403.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09635"}},"official":{"repos":["akhilkedia/tranformersgetstable"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dam-dynamic-adapter-merging-for-continual","slug":"dam-dynamic-adapter-merging-for-continual","title":"DAM: Dynamic Adapter Merging for Continual Video QA Learning","date":"2024-03-13","arxiv_id":"2403.08755","repositories_listed":1,"syntology":null},{"url":"/paper/moleculeqa-a-dataset-to-evaluate-factual","slug":"moleculeqa-a-dataset-to-evaluate-factual","title":"MoleculeQA: A Dataset to Evaluate Factual Accuracy in Molecular Comprehension","date":"2024-03-13","arxiv_id":"2403.08192","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-memorization-the-challenge-of-random","slug":"beyond-memorization-the-challenge-of-random","title":"Beyond Memorization: The Challenge of Random Memory Access in Language Models","date":"2024-03-12","arxiv_id":"2403.07805","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-memorization-the-challenge-of-random#ran","syntology_url":"https://syntology.ai/paper/2403.07805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07805"}},"official":{"repos":["sail-sg/lm-random-memory-access"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/branch-train-mix-mixing-expert-llms-into-a","slug":"branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","arxiv_id":"2403.07816","repositories_listed":1,"syntology":null},{"url":"/paper/complex-reasoning-over-logical-queries-on","slug":"complex-reasoning-over-logical-queries-on","title":"Complex Reasoning over Logical Queries on Commonsense Knowledge Graphs","date":"2024-03-12","arxiv_id":"2403.07398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/complex-reasoning-over-logical-queries-on#ran","syntology_url":"https://syntology.ai/paper/2403.07398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07398"}},"official":{"repos":["tqfang/complex-commonsense-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/alarm-align-language-models-via-hierarchical","slug":"alarm-align-language-models-via-hierarchical","title":"ALaRM: Align Language Models via Hierarchical Rewards Modeling","date":"2024-03-11","arxiv_id":"2403.06754","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/alarm-align-language-models-via-hierarchical#ran","syntology_url":"https://syntology.ai/paper/2403.06754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06754"}},"official":{"repos":["halfrot/ALaRM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/answering-diverse-questions-via-text-attached","slug":"answering-diverse-questions-via-text-attached","title":"Answering Diverse Questions via Text Attached with Key Audio-Visual Clues","date":"2024-03-11","arxiv_id":"2403.06679","repositories_listed":1,"syntology":null},{"url":"/paper/era-cot-improving-chain-of-thought-through","slug":"era-cot-improving-chain-of-thought-through","title":"ERA-CoT: Improving Chain-of-Thought through Entity Relationship Analysis","date":"2024-03-11","arxiv_id":"2403.06932","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/era-cot-improving-chain-of-thought-through#ran","syntology_url":"https://syntology.ai/paper/2403.06932","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06932"}},"official":{"repos":["oceanntwt/era-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-towards-a-computational-friendly-cloud","slug":"spa-towards-a-computational-friendly-cloud","title":"SPA: Towards A Computational Friendly Cloud-Base and On-Devices Collaboration Seq2seq Personalized Generation with Casual Inference","date":"2024-03-11","arxiv_id":"2403.07088","repositories_listed":1,"syntology":null},{"url":"/paper/calibrating-large-language-models-using-their","slug":"calibrating-large-language-models-using-their","title":"Calibrating Large Language Models Using Their Generations Only","date":"2024-03-09","arxiv_id":"2403.05973","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calibrating-large-language-models-using-their#ran","syntology_url":"https://syntology.ai/paper/2403.05973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05973"}},"official":{"repos":["parameterlab/apricot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kg-rank-enhancing-large-language-models-for","slug":"kg-rank-enhancing-large-language-models-for","title":"KG-Rank: Enhancing Large Language Models for Medical QA with Knowledge Graphs and Ranking Techniques","date":"2024-03-09","arxiv_id":"2403.05881","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kg-rank-enhancing-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2403.05881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05881"}},"official":{"repos":["yangrui525/kg-rank"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bias-augmented-consistency-training-reduces","slug":"bias-augmented-consistency-training-reduces","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","date":"2024-03-08","arxiv_id":"2403.05518","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bias-augmented-consistency-training-reduces#ran","syntology_url":"https://syntology.ai/paper/2403.05518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05518"}},"official":{"repos":["raybears/cot-transparency"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-t-remember-details-in-long-documents-you","slug":"can-t-remember-details-in-long-documents-you","title":"Can't Remember Details in Long Documents? You Need Some R&R","date":"2024-03-08","arxiv_id":"2403.05004","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-t-remember-details-in-long-documents-you#ran","syntology_url":"https://syntology.ai/paper/2403.05004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05004"}},"official":{"repos":["casetext/r-and-r"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/debiasing-large-visual-language-models","slug":"debiasing-large-visual-language-models","title":"Debiasing Multimodal Large Language Models","date":"2024-03-08","arxiv_id":"2403.05262","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/debiasing-large-visual-language-models#ran","syntology_url":"https://syntology.ai/paper/2403.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05262"}},"official":{"repos":["yfzhang114/llava-align"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/gemini-1-5-unlocking-multimodal-understanding","slug":"gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","arxiv_id":"2403.05530","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-multi-role-capabilities-of-large","slug":"harnessing-multi-role-capabilities-of-large","title":"Harnessing Multi-Role Capabilities of Large Language Models for Open-Domain Question Answering","date":"2024-03-08","arxiv_id":"2403.05217","repositories_listed":1,"syntology":null},{"url":"/paper/cat-enhancing-multimodal-large-language-model","slug":"cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","arxiv_id":"2403.04640","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cat-enhancing-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04640"}},"official":{"repos":["rikeilong/bay-cat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-chain-of-thought-driven-reasoning-to","slug":"few-shot-chain-of-thought-driven-reasoning-to","title":"Few shot chain-of-thought driven reasoning to prompt LLMs for open ended medical question answering","date":"2024-03-07","arxiv_id":"2403.04890","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/few-shot-chain-of-thought-driven-reasoning-to#ran","syntology_url":"https://syntology.ai/paper/2403.04890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04890"}},"official":{"repos":["coldseal/clinicr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/halueval-wild-evaluating-hallucinations-of","slug":"halueval-wild-evaluating-hallucinations-of","title":"HaluEval-Wild: Evaluating Hallucinations of Language Models in the Wild","date":"2024-03-07","arxiv_id":"2403.04307","repositories_listed":1,"syntology":null},{"url":"/paper/qaq-quality-adaptive-quantization-for-llm-kv","slug":"qaq-quality-adaptive-quantization-for-llm-kv","title":"QAQ: Quality Adaptive Quantization for LLM KV Cache","date":"2024-03-07","arxiv_id":"2403.04643","repositories_listed":1,"syntology":null},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-hallucination-in-large-language","slug":"benchmarking-hallucination-in-large-language","title":"Benchmarking Hallucination in Large Language Models based on Unanswerable Math Word Problem","date":"2024-03-06","arxiv_id":"2403.03558","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-elementary-multilingual","slug":"evaluating-the-elementary-multilingual","title":"Evaluating the Elementary Multilingual Capabilities of Large Language Models with MultiQ","date":"2024-03-06","arxiv_id":"2403.03814","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-the-elementary-multilingual#ran","syntology_url":"https://syntology.ai/paper/2403.03814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03814"}},"official":{"repos":["paul-rottger/multiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evidence-focused-fact-summarization-for","slug":"evidence-focused-fact-summarization-for","title":"Evidence-Focused Fact Summarization for Knowledge-Augmented Zero-Shot Question Answering","date":"2024-03-05","arxiv_id":"2403.02966","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/evidence-focused-fact-summarization-for#ran","syntology_url":"https://syntology.ai/paper/2403.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02966"}},"official":{"repos":["anon809/efsum"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/brilla-ai-ai-contestant-for-the-national","slug":"brilla-ai-ai-contestant-for-the-national","title":"Brilla AI: AI Contestant for the National Science and Maths Quiz","date":"2024-03-04","arxiv_id":"2403.01699","repositories_listed":1,"syntology":null},{"url":"/paper/eee-qa-exploring-effective-and-efficient","slug":"eee-qa-exploring-effective-and-efficient","title":"EEE-QA: Exploring Effective and Efficient Question-Answer Representations","date":"2024-03-04","arxiv_id":"2403.02176","repositories_listed":1,"syntology":null},{"url":"/paper/to-generate-or-to-retrieve-on-the","slug":"to-generate-or-to-retrieve-on-the","title":"To Generate or to Retrieve? On the Effectiveness of Artificial Contexts for Medical Open-Domain Question Answering","date":"2024-03-04","arxiv_id":"2403.01924","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-for-medical-report","slug":"vision-language-models-for-medical-report","title":"Vision-Language Models for Medical Report Generation and Visual Question Answering: A Review","date":"2024-03-04","arxiv_id":"2403.02469","repositories_listed":1,"syntology":null},{"url":"/paper/cr-lt-kgqa-a-knowledge-graph-question","slug":"cr-lt-kgqa-a-knowledge-graph-question","title":"CR-LT-KGQA: A Knowledge Graph Question Answering Dataset Requiring Commonsense Reasoning and Long-Tail Knowledge","date":"2024-03-03","arxiv_id":"2403.01395","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cr-lt-kgqa-a-knowledge-graph-question#ran","syntology_url":"https://syntology.ai/paper/2403.01395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01395"}},"official":{"repos":["d3mlab/cr-lt-kgqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-vs-retrieval-augmented-generation","slug":"fine-tuning-vs-retrieval-augmented-generation","title":"Fine Tuning vs. Retrieval Augmented Generation for Less Popular Knowledge","date":"2024-03-03","arxiv_id":"2403.01432","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fine-tuning-vs-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2403.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01432"}},"official":{"repos":["heydarsoudani/ragvsft"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/syllabusqa-a-course-logistics-question","slug":"syllabusqa-a-course-logistics-question","title":"SyllabusQA: A Course Logistics Question Answering Dataset","date":"2024-03-03","arxiv_id":"2403.14666","repositories_listed":1,"syntology":null},{"url":"/paper/localrqa-from-generating-data-to-locally","slug":"localrqa-from-generating-data-to-locally","title":"LocalRQA: From Generating Data to Locally Training, Testing, and Deploying Retrieval-Augmented QA Systems","date":"2024-03-01","arxiv_id":"2403.00982","repositories_listed":1,"syntology":null},{"url":"/paper/let-llms-take-on-the-latest-challenges-a","slug":"let-llms-take-on-the-latest-challenges-a","title":"Let LLMs Take on the Latest Challenges! A Chinese Dynamic Question Answering Benchmark","date":"2024-02-29","arxiv_id":"2402.19248","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-llms-take-on-the-latest-challenges-a#ran","syntology_url":"https://syntology.ai/paper/2402.19248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19248"}},"official":{"repos":["alibaba-nlp/cdqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-truthfulness-in-large-language","slug":"characterizing-truthfulness-in-large-language","title":"Characterizing Truthfulness in Large Language Model Generations with Local Intrinsic Dimension","date":"2024-02-28","arxiv_id":"2402.18048","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/characterizing-truthfulness-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.18048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18048"}},"official":{"repos":["fanyin3639/lid-hallucinationdetection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-first-place-solution-of-wsdm-cup-2024","slug":"the-first-place-solution-of-wsdm-cup-2024","title":"The First Place Solution of WSDM Cup 2024: Leveraging Large Language Models for Conversational Multi-Doc QA","date":"2024-02-28","arxiv_id":"2402.18385","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-information-refinement-training","slug":"unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","arxiv_id":"2402.18150","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-information-refinement-training#ran","syntology_url":"https://syntology.ai/paper/2402.18150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18150"}},"official":{"repos":["xsc1234/info-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/blendsql-a-scalable-dialect-for-unifying","slug":"blendsql-a-scalable-dialect-for-unifying","title":"BlendSQL: A Scalable Dialect for Unifying Hybrid Question Answering in Relational Algebra","date":"2024-02-27","arxiv_id":"2402.17882","repositories_listed":1,"syntology":null},{"url":"/paper/can-llm-generate-culturally-relevant","slug":"can-llm-generate-culturally-relevant","title":"Can LLM Generate Culturally Relevant Commonsense QA Data? Case Study in Indonesian and Sundanese","date":"2024-02-27","arxiv_id":"2402.17302","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llm-generate-culturally-relevant#ran","syntology_url":"https://syntology.ai/paper/2402.17302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17302"}},"official":{"repos":["rifkiaputri/id-csqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-very-long-term-conversational","slug":"evaluating-very-long-term-conversational","title":"Evaluating Very Long-Term Conversational Memory of LLM Agents","date":"2024-02-27","arxiv_id":"2402.17753","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/evaluating-very-long-term-conversational#ran","syntology_url":"https://syntology.ai/paper/2402.17753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17753"}},"official":null}},{"url":"/paper/fact-and-reflection-far-improves-confidence","slug":"fact-and-reflection-far-improves-confidence","title":"Fact-and-Reflection (FaR) Improves Confidence Calibration of Large Language Models","date":"2024-02-27","arxiv_id":"2402.17124","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fact-and-reflection-far-improves-confidence#ran","syntology_url":"https://syntology.ai/paper/2402.17124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17124"}},"official":{"repos":["colinzhaoust/fact-and-reflection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/jmlr-joint-medical-llm-and-retrieval-training","slug":"jmlr-joint-medical-llm-and-retrieval-training","title":"JMLR: Joint Medical LLM and Retrieval Training for Enhancing Reasoning and Professional Question Answering Capability","date":"2024-02-27","arxiv_id":"2402.17887","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-on-tabular-data-a","slug":"large-language-models-on-tabular-data-a","title":"Large Language Models(LLMs) on Tabular Data: Prediction, Generation, and Understanding -- A Survey","date":"2024-02-27","arxiv_id":"2402.17944","repositories_listed":1,"syntology":null},{"url":"/paper/mathsensei-a-tool-augmented-large-language","slug":"mathsensei-a-tool-augmented-large-language","title":"MATHSENSEI: A Tool-Augmented Large Language Model for Mathematical Reasoning","date":"2024-02-27","arxiv_id":"2402.17231","repositories_listed":1,"syntology":null},{"url":"/paper/nextlevelbert-investigating-masked-language","slug":"nextlevelbert-investigating-masked-language","title":"NextLevelBERT: Masked Language Modeling with Higher-Level Representations for Long Documents","date":"2024-02-27","arxiv_id":"2402.17682","repositories_listed":1,"syntology":null},{"url":"/paper/rear-a-relevance-aware-retrieval-augmented","slug":"rear-a-relevance-aware-retrieval-augmented","title":"REAR: A Relevance-Aware Retrieval-Augmented Framework for Open-Domain Question Answering","date":"2024-02-27","arxiv_id":"2402.17497","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":14,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/rear-a-relevance-aware-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2402.17497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17497"}},"official":{"repos":["rucaibox/rear"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/truthx-alleviating-hallucinations-by-editing","slug":"truthx-alleviating-hallucinations-by-editing","title":"TruthX: Alleviating Hallucinations by Editing Large Language Models in Truthful Space","date":"2024-02-27","arxiv_id":"2402.17811","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/truthx-alleviating-hallucinations-by-editing#ran","syntology_url":"https://syntology.ai/paper/2402.17811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17811"}},"official":{"repos":["ictnlp/truthx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-assisted-multi-teacher-continual-learning","slug":"llm-assisted-multi-teacher-continual-learning","title":"LLM-Assisted Multi-Teacher Continual Learning for Visual Question Answering in Robotic Surgery","date":"2024-02-26","arxiv_id":"2402.16664","repositories_listed":1,"syntology":null},{"url":"/paper/mozip-a-multilingual-benchmark-to-evaluate","slug":"mozip-a-multilingual-benchmark-to-evaluate","title":"MoZIP: A Multilingual Benchmark to Evaluate Large Language Models in Intellectual Property","date":"2024-02-26","arxiv_id":"2402.16389","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-cross-lingual-open-domain","slug":"pre-training-cross-lingual-open-domain","title":"Pre-training Cross-lingual Open Domain Question Answering with Large-scale Synthetic Supervision","date":"2024-02-26","arxiv_id":"2402.16508","repositories_listed":1,"syntology":null},{"url":"/paper/retrievalqa-assessing-adaptive-retrieval","slug":"retrievalqa-assessing-adaptive-retrieval","title":"RetrievalQA: Assessing Adaptive Retrieval-Augmented Generation for Short-form Open-Domain Question Answering","date":"2024-02-26","arxiv_id":"2402.16457","repositories_listed":1,"syntology":null},{"url":"/paper/ehrnoteqa-a-patient-specific-question","slug":"ehrnoteqa-a-patient-specific-question","title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","date":"2024-02-25","arxiv_id":"2402.16040","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrnoteqa-a-patient-specific-question#ran","syntology_url":"https://syntology.ai/paper/2402.16040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16040"}},"official":{"repos":["ji-youn-kim/ehrnoteqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedical-entity-linking-as-multiple-choice","slug":"biomedical-entity-linking-as-multiple-choice","title":"Biomedical Entity Linking as Multiple Choice Question Answering","date":"2024-02-23","arxiv_id":"2402.15189","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-kbqa-multi-turn-interactions-for","slug":"interactive-kbqa-multi-turn-interactions-for","title":"Interactive-KBQA: Multi-Turn Interactions for Knowledge Base Question Answering with Large Language Models","date":"2024-02-23","arxiv_id":"2402.15131","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":3,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/interactive-kbqa-multi-turn-interactions-for#ran","syntology_url":"https://syntology.ai/paper/2402.15131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15131"}},"official":{"repos":["jimxionggm/interactive-kbqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/commvqa-situating-visual-question-answering","slug":"commvqa-situating-visual-question-answering","title":"CommVQA: Situating Visual Question Answering in Communicative Contexts","date":"2024-02-22","arxiv_id":"2402.15002","repositories_listed":1,"syntology":null},{"url":"/paper/data-science-with-llms-and-interpretable","slug":"data-science-with-llms-and-interpretable","title":"Data Science with LLMs and Interpretable Models","date":"2024-02-22","arxiv_id":"2402.14474","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-implicitly-determine-the-suitable","slug":"do-llms-implicitly-determine-the-suitable","title":"Do LLMs Implicitly Determine the Suitable Text Difficulty for Users?","date":"2024-02-22","arxiv_id":"2402.14453","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-large-language-models-for-concept","slug":"leveraging-large-language-models-for-concept","title":"Leveraging Large Language Models for Concept Graph Recovery and Question Answering in NLP Education","date":"2024-02-22","arxiv_id":"2402.14293","repositories_listed":1,"syntology":null},{"url":"/paper/simplot-enhancing-chart-question-answering-by","slug":"simplot-enhancing-chart-question-answering-by","title":"SIMPLOT: Enhancing Chart Question Answering by Distilling Essentials","date":"2024-02-22","arxiv_id":"2405.00021","repositories_listed":1,"syntology":null},{"url":"/paper/triad-a-framework-leveraging-a-multi-role-llm","slug":"triad-a-framework-leveraging-a-multi-role-llm","title":"Triad: A Framework Leveraging a Multi-Role LLM-based Agent to Solve Knowledge Base Question Answering","date":"2024-02-22","arxiv_id":"2402.14320","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/triad-a-framework-leveraging-a-multi-role-llm#ran","syntology_url":"https://syntology.ai/paper/2402.14320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14320"}},"official":{"repos":["ZJU-DCDLab/Triad"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-hallucinations-of-multi-modal-large","slug":"visual-hallucinations-of-multi-modal-large","title":"Visual Hallucinations of Multi-modal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14683","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-hallucinations-of-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2402.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14683"}},"official":{"repos":["wenhuang2000/vhtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/activerag-revealing-the-treasures-of","slug":"activerag-revealing-the-treasures-of","title":"ActiveRAG: Autonomously Knowledge Assimilation and Accommodation through Retrieval-Augmented Agents","date":"2024-02-21","arxiv_id":"2402.13547","repositories_listed":1,"syntology":null},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fanoutqa-multi-hop-multi-document-question","slug":"fanoutqa-multi-hop-multi-document-question","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.14116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fanoutqa-multi-hop-multi-document-question#ran","syntology_url":"https://syntology.ai/paper/2402.14116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14116"}},"official":{"repos":["zhudotexe/fanoutqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-poison-large-language-models","slug":"learning-to-poison-large-language-models","title":"Learning to Poison Large Language Models for Downstream Manipulation","date":"2024-02-21","arxiv_id":"2402.13459","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-to-poison-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.13459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13459"}},"official":{"repos":["rookiezxy/gbtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pqa-zero-shot-protein-question-answering-for","slug":"pqa-zero-shot-protein-question-answering-for","title":"PQA: Zero-shot Protein Question Answering for Free-form Scientific Enquiry with Large Language Models","date":"2024-02-21","arxiv_id":"2402.13653","repositories_listed":1,"syntology":null},{"url":"/paper/refutebench-evaluating-refuting-instruction","slug":"refutebench-evaluating-refuting-instruction","title":"RefuteBench: Evaluating Refuting Instruction-Following for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13463","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-helps-or-hurts-a-deeper-dive-into","slug":"retrieval-helps-or-hurts-a-deeper-dive-into","title":"Retrieval Helps or Hurts? A Deeper Dive into the Efficacy of Retrieval Augmentation to Language Models","date":"2024-02-21","arxiv_id":"2402.13492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-helps-or-hurts-a-deeper-dive-into#ran","syntology_url":"https://syntology.ai/paper/2402.13492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13492"}},"official":{"repos":["megagonlabs/witqa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/towards-building-multilingual-language-model","slug":"towards-building-multilingual-language-model","title":"Towards Building Multilingual Language Model for Medicine","date":"2024-02-21","arxiv_id":"2402.13963","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-building-multilingual-language-model#ran","syntology_url":"https://syntology.ai/paper/2402.13963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13963"}},"official":{"repos":["magic-ai4med/mmedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bimedix-bilingual-medical-mixture-of-experts","slug":"bimedix-bilingual-medical-mixture-of-experts","title":"BiMediX: Bilingual Medical Mixture of Experts LLM","date":"2024-02-20","arxiv_id":"2402.13253","repositories_listed":1,"syntology":null},{"url":"/paper/drbenchmark-a-large-language-understanding","slug":"drbenchmark-a-large-language-understanding","title":"DrBenchmark: A Large Language Understanding Evaluation Benchmark for French Biomedical Domain","date":"2024-02-20","arxiv_id":"2402.13432","repositories_listed":1,"syntology":null}],"record_sha256":"2c0cb4cafe2c2a8096a32fc88196f864057962132a804b052b73d186beb3a657","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}