{"url":"/task/multi-hop-question-answering","name":"Multi-hop Question Answering","slug":"multi-hop-question-answering","description_markdown":null,"categories":[{"name":"Knowledge Base","url":"/area/knowledge-base"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":202,"papers_with_code":92,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/multi-hop-question-answering-on-concurrentqa","slug":"multi-hop-question-answering-on-concurrentqa","dataset":"ConcurrentQA","dataset_url":"/dataset/simran-arora","rows_in_archive":1,"metrics":["Answer F1"],"first_row_in_archive_order":{"model":"Multi-hop Dense Passage Retriever (MDR)","paper_title":"Reasoning over Public and Private Data in Retrieval-Based Systems","paper_url":"/paper/reasoning-over-public-and-private-data-in","paper_date":"2022-03-14","arxiv_id":"2203.11027","code_links":[{"title":"facebookresearch/concurrentqa","url":"https://github.com/facebookresearch/concurrentqa"}],"syntology":null}},{"leaderboard":"/sota/multi-hop-question-answering-on-musique-ans","slug":"multi-hop-question-answering-on-musique-ans","dataset":"MuSiQue-Ans","dataset_url":"/dataset/musique-ans","rows_in_archive":1,"metrics":["An","Sp"],"first_row_in_archive_order":{"model":"Beam Retrieval","paper_title":"End-to-End Beam Retrieval for Multi-Hop Question Answering","paper_url":"/paper/beam-retrieval-general-end-to-end-retrieval","paper_date":"2023-08-17","arxiv_id":"2308.08973","code_links":[{"title":"ShayekhBinIslam/openrag","url":"https://github.com/ShayekhBinIslam/openrag"},{"title":"canghongjian/beam_retriever","url":"https://github.com/canghongjian/beam_retriever"},{"title":"Alab-NII/2wikimultihop","url":"https://github.com/Alab-NII/2wikimultihop"}],"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/musique-ans","name":"MuSiQue-Ans","full_name":"","num_papers_in_archive":10},{"url":"/dataset/spartqa-1","name":"SPARTQA -","full_name":"SPAtial Reasoning on Textual Question Answering.","num_papers_in_archive":9},{"url":"/dataset/simran-arora","name":"ConcurrentQA Benchmark","full_name":"","num_papers_in_archive":3},{"url":"/dataset/multiq","name":"MultiQ","full_name":"MultiQ","num_papers_in_archive":3}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":92,"tagged_in_all":202,"items":[{"url":"/paper/qa-gnn-reasoning-with-language-models-and","title":"QA-GNN: Reasoning with Language Models and Knowledge Graphs for Question Answering","date":"2021-04-13","arxiv_id":"2104.06378","repositories_listed":6,"syntology":{"n":25,"n_ran":1,"n_unverified":24,"n_pointer_only":4}},{"url":"/paper/190409380","title":"Repurposing Entailment for Multi-Hop Question Answering Tasks","date":"2019-04-20","arxiv_id":"1904.09380","repositories_listed":4,"syntology":null},{"url":"/paper/beam-retrieval-general-end-to-end-retrieval","title":"End-to-End Beam Retrieval for Multi-Hop Question Answering","date":"2023-08-17","arxiv_id":"2308.08973","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/musique-multi-hop-questions-via-single-hop","title":"MuSiQue: Multihop Questions via Single-hop Question Composition","date":"2021-08-02","arxiv_id":"2108.00573","repositories_listed":3,"syntology":{"n":16,"n_ran":1,"n_unverified":15,"n_pointer_only":1}},{"url":"/paper/multi-hop-question-answering-via-reasoning","title":"Multi-hop Question Answering via Reasoning Chains","date":"2019-10-07","arxiv_id":"1910.02610","repositories_listed":3,"syntology":null},{"url":"/paper/hipporag-neurobiologically-inspired-long-term","title":"HippoRAG: Neurobiologically Inspired Long-Term Memory for Large Language Models","date":"2024-05-23","arxiv_id":"2405.14831","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/deepedit-knowledge-editing-as-decoding-with","title":"DeepEdit: Knowledge Editing as Decoding with Constraints","date":"2024-01-19","arxiv_id":"2401.10471","repositories_listed":2,"syntology":null},{"url":"/paper/mquake-assessing-knowledge-editing-in","title":"MQuAKE: Assessing Knowledge Editing in Language Models via Multi-Hop Questions","date":"2023-05-24","arxiv_id":"2305.14795","repositories_listed":2,"syntology":null},{"url":"/paper/can-language-models-solve-graph-problems-in","title":"Can Language Models Solve Graph Problems in Natural Language?","date":"2023-05-17","arxiv_id":"2305.10037","repositories_listed":2,"syntology":{"n":22,"n_ran":2,"n_unverified":20,"n_pointer_only":0}},{"url":"/paper/analyzing-the-effectiveness-of-the-underlying","title":"Analyzing the Effectiveness of the Underlying Reasoning Tasks in Multi-hop Question Answering","date":"2023-02-12","arxiv_id":"2302.05963","repositories_listed":2,"syntology":null},{"url":"/paper/rethinking-label-smoothing-on-multi-hop","title":"Rethinking Label Smoothing on Multi-hop Question Answering","date":"2022-12-19","arxiv_id":"2212.09512","repositories_listed":2,"syntology":null},{"url":"/paper/improving-multi-hop-question-answering-over","title":"Improving Multi-hop Question Answering over Knowledge Graphs using Knowledge Base Embeddings","date":"2020-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/hybridqa-a-dataset-of-multi-hop-question","title":"HybridQA: A Dataset of Multi-Hop Question Answering over Tabular and Textual Data","date":"2020-04-15","arxiv_id":"2004.07347","repositories_listed":2,"syntology":{"n":15,"n_ran":5,"n_unverified":10,"n_pointer_only":2}},{"url":"/paper/cognitive-graph-for-multi-hop-reading","title":"Cognitive Graph for Multi-Hop Reading Comprehension at Scale","date":"2019-05-14","arxiv_id":"1905.05460","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/hotpotqa-a-dataset-for-diverse-explainable","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","date":"2018-09-25","arxiv_id":"1809.09600","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/commonsense-for-generative-multi-hop-question","title":"Commonsense for Generative Multi-Hop Question Answering Tasks","date":"2018-09-17","arxiv_id":"1809.06309","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/dynamic-chunking-and-selection-for-reading","title":"Dynamic Chunking and Selection for Reading Comprehension of Ultra-Long Context in Large Language Models","date":"2025-06-01","arxiv_id":"2506.00773","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":9}},{"url":"/paper/scaling-external-knowledge-input-beyond","title":"Scaling External Knowledge Input Beyond Context Windows of LLMs via Multi-Agent Collaboration","date":"2025-05-27","arxiv_id":"2505.21471","repositories_listed":1,"syntology":null},{"url":"/paper/masksearch-a-universal-pre-training-framework","title":"MASKSEARCH: A Universal Pre-Training Framework to Enhance Agentic Search Capability","date":"2025-05-26","arxiv_id":"2505.20285","repositories_listed":1,"syntology":null},{"url":"/paper/knowtrace-bootstrapping-iterative-retrieval","title":"KnowTrace: Bootstrapping Iterative Retrieval-Augmented Generation with Structured Knowledge Tracing","date":"2025-05-26","arxiv_id":"2505.20245","repositories_listed":1,"syntology":null},{"url":"/paper/self-critique-guided-iterative-reasoning-for","title":"Self-Critique Guided Iterative Reasoning for Multi-hop Question Answering","date":"2025-05-25","arxiv_id":"2505.19112","repositories_listed":1,"syntology":null},{"url":"/paper/situatedthinker-grounding-llm-reasoning-with","title":"SituatedThinker: Grounding LLM Reasoning with Real-World through Situated Thinking","date":"2025-05-25","arxiv_id":"2505.19300","repositories_listed":1,"syntology":null},{"url":"/paper/hopweaver-synthesizing-authentic-multi-hop","title":"HopWeaver: Synthesizing Authentic Multi-Hop Questions Across Text Corpora","date":"2025-05-21","arxiv_id":"2505.15087","repositories_listed":1,"syntology":null},{"url":"/paper/masking-in-multi-hop-qa-an-analysis-of-how","title":"Masking in Multi-hop QA: An Analysis of How Language Models Perform with Context Permutation","date":"2025-05-16","arxiv_id":"2505.11754","repositories_listed":1,"syntology":null},{"url":"/paper/focus-merge-rank-improved-question-answering","title":"Focus, Merge, Rank: Improved Question Answering Based on Semi-structured Knowledge Bases","date":"2025-05-14","arxiv_id":"2505.09246","repositories_listed":1,"syntology":null},{"url":"/paper/treehop-generate-and-filter-next-query","title":"TreeHop: Generate and Filter Next Query Embeddings Efficiently for Multi-hop Question Answering","date":"2025-04-28","arxiv_id":"2504.20114","repositories_listed":1,"syntology":null},{"url":"/paper/collab-rag-boosting-retrieval-augmented","title":"Collab-RAG: Boosting Retrieval-Augmented Generation for Complex Question Answering via White-Box and Black-Box LLM Collaboration","date":"2025-04-07","arxiv_id":"2504.04915","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/retrieval-augmented-generation-with-1","title":"Retrieval-Augmented Generation with Hierarchical Knowledge","date":"2025-03-13","arxiv_id":"2503.10150","repositories_listed":1,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/levelrag-enhancing-retrieval-augmented","title":"LevelRAG: Enhancing Retrieval-Augmented Generation with Multi-hop Logic Planning over Rewriting Augmented Searchers","date":"2025-02-25","arxiv_id":"2502.18139","repositories_listed":1,"syntology":null},{"url":"/paper/mintqa-a-multi-hop-question-answering","title":"MINTQA: A Multi-Hop Question Answering Benchmark for Evaluating LLMs on New and Tail Knowledge","date":"2024-12-22","arxiv_id":"2412.17032","repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}