{"url":"/dataset/triviaqa","name":"TriviaQA","full_name":null,"description_markdown":"**TriviaQA** is a realistic text-based question answering dataset which includes 950K question-answer pairs from 662K documents collected from Wikipedia and the web. This dataset is more challenging than standard QA benchmark datasets such as Stanford Question Answering Dataset (SQuAD), as the answers for a question may not be directly obtained by span prediction and the context is very long. TriviaQA dataset consists of both human-verified and machine-generated QA subsets.\r\n\r\nSource: [Episodic Memory Reader: Learning What to Rememberfor Question Answering from Streaming Data](https://arxiv.org/abs/1903.06164)\r\nImage Source: [Joshi et al](https://arxiv.org/pdf/1705.03551v2.pdf)","description_withheld":null,"homepage":"http://nlp.cs.washington.edu/triviaqa/","introduced_date":"2017-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/triviaqa-a-large-scale-distantly-supervised","title":"TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension","first_author":"Mandar Joshi","url":null},"license":{"name":"Unknown","url":"http://nlp.cs.washington.edu/triviaqa/#:~:text=copyright"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"},{"name":"Reading Comprehension","url":"/task/reading-comprehension","datasets_with_task":"/datasets/task/reading-comprehension"},{"name":"Open-Domain Question Answering","url":"/task/open-domain-question-answering","datasets_with_task":"/datasets/task/open-domain-question-answering"},{"name":"Question Generation","url":"/task/question-generation","datasets_with_task":"/datasets/task/question-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["TriviaQA","KILT: TriviaQA"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/trivia_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/mandarjoshi/trivia_qa","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/facebookresearch/ParlAI","url":"https://parl.ai/docs/tasks.html#triviaqa","frameworks":["pytorch"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/trivia_qa","frameworks":["tf","jax"]},{"repo":"https://github.com/RUCAIBox/LLMBox","url":"https://github.com/RUCAIBox/LLMBox","frameworks":["pytorch"]},{"repo":"https://github.com/allenai/allennlp-models","url":"https://docs.allennlp.org/models/main/models/rc/dataset_readers/triviaqa/","frameworks":["pytorch"]}],"num_papers_in_archive":953,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-triviaqa","task":"Question Answering","dataset_variant":"TriviaQA","rows":56,"metrics":["EM","F1"],"first_row_in_archive_order":{"model":"Claude 2 (few-shot, k=5)","paper":"/paper/model-card-and-evaluations-for-claude-models","metrics":{"EM":"87.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-domain-question-answering-on-kilt-2","task":"Open-Domain Question Answering","dataset_variant":"KILT: TriviaQA","rows":15,"metrics":["KILT-EM","R-Prec","Recall@5","EM","F1","KILT-F1"],"first_row_in_archive_order":{"model":"Re2G","paper":"/paper/re2g-retrieve-rerank-generate-2","metrics":{"EM":"76.27","F1":"81.4","KILT-EM":"57.91","KILT-F1":"61.78","R-Prec":"72.68","Recall@5":"74.23"},"code_links":[{"title":"ibm/kgi-slot-filling","url":"https://github.com/ibm/kgi-slot-filling"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/question-generation-on-triviaqa","task":"Question Generation","dataset_variant":"TriviaQA","rows":2,"metrics":["QAE","R-QAE"],"first_row_in_archive_order":{"model":"Info-HCVAE","paper":"/paper/generating-diverse-and-consistent-qa-pairs","metrics":{"QAE":"35.45","R-QAE":"21.65"},"code_links":[{"title":"seanie12/Info-HCVAE","url":"https://github.com/seanie12/Info-HCVAE"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/open-domain-question-answering-on-triviaqa","task":"Open-Domain Question Answering","dataset_variant":"TriviaQA","rows":1,"metrics":["Exact Match"],"first_row_in_archive_order":{"model":"UnitedQA (Hybrid)","paper":"/paper/unitedqa-a-hybrid-approach-for-open-domain","metrics":{"Exact Match":"70.5"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/text-generation-on-triviaqa","task":"Text Generation","dataset_variant":"TriviaQA","rows":0,"metrics":["acc"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/search-o1-agentic-search-enhanced-large","title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models","date":"2025-01-09","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/shakti-a-2-5-billion-parameter-small-language","title":"SHAKTI: A 2.5 Billion Parameter Small Language Model Optimized for Edge AI and Low-Resource Environments","date":"2024-10-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/rankrag-unifying-context-ranking-with","title":"RankRAG: Unifying Context Ranking with Retrieval-Augmented Generation in LLMs","date":"2024-07-02","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":4,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/branch-train-mix-mixing-expert-llms-into-a","title":"Branch-Train-MiX: Mixing Expert LLMs into a Mixture-of-Experts LLM","date":"2024-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/chatqa-building-gpt-4-level-conversational-qa","title":"ChatQA: Surpassing GPT-4 on Conversational QA and RAG","date":"2024-01-18","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":9,"samples_unverified":2,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/ra-dit-retrieval-augmented-dual-instruction","title":"RA-DIT: Retrieval-Augmented Dual Instruction Tuning","date":"2023-10-02","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/llama-2-open-foundation-and-fine-tuned-chat","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","date":"2023-07-18","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":52,"samples_ran":31,"samples_unverified":21,"pointer_only_for_licence":16,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/palm-2-technical-report-1","title":"PaLM 2 Technical Report","date":"2023-05-17","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":2,"samples_unverified":3,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/llama-open-and-efficient-foundation-language-1","title":"LLaMA: Open and Efficient Foundation Language Models","date":"2023-02-27","rows_on_this_dataset":4,"code_links":57,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":58,"samples_ran":26,"samples_unverified":32,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/replug-retrieval-augmented-black-box-language","title":"REPLUG: Retrieval-Augmented Black-Box Language Models","date":"2023-01-30","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":13,"samples_ran":0,"samples_unverified":13,"pointer_only_for_licence":13,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/fie-building-a-global-probability-space-by","title":"FiE: Building a Global Probability Space by Leveraging Early Fusion in Encoder for Open-Domain Question Answering","date":"2022-11-18","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dyrex-dynamic-query-representation-for","title":"DyREx: Dynamic Query Representation for Extractive Question Answering","date":"2022-10-26","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":2,"samples_unverified":1,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/re2g-retrieve-rerank-generate-2","title":"Re2G: Retrieve, Rerank, Generate","date":"2022-07-13","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":8,"samples_ran":1,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","rows_on_this_dataset":3,"code_links":7,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":37,"samples_ran":30,"samples_unverified":7,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/linkbert-pretraining-language-models-with","title":"LinkBERT: Pretraining Language Models with Document Links","date":"2022-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":0,"samples_unverified":14,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/glam-efficient-scaling-of-language-models","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","date":"2021-12-13","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/mention-memory-incorporating-textual-1","title":"Mention Memory: incorporating textual knowledge into Transformers through entity mention attention","date":"2021-10-12","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/reasonbert-pre-trained-to-reason-with-distant","title":"ReasonBERT: Pre-trained to Reason with Distant Supervision","date":"2021-09-10","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":11,"samples_ran":3,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/end-to-end-training-of-multi-document-reader","title":"End-to-End Training of Multi-Document Reader and Retriever for Open-Domain Question Answering","date":"2021-06-09","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":4,"samples_unverified":0,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/unitedqa-a-hybrid-approach-for-open-domain","title":"UnitedQA: A Hybrid Approach for Open Domain Question Answering","date":"2021-01-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/distilling-knowledge-from-reader-to-retriever-1","title":"Distilling Knowledge from Reader to Retriever for Question Answering","date":"2020-12-08","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/kilt-a-benchmark-for-knowledge-intensive","title":"KILT: a Benchmark for Knowledge Intensive Language Tasks","date":"2020-09-04","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":1,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/big-bird-transformers-for-longer-sequences","title":"Big Bird: Transformers for Longer Sequences","date":"2020-07-28","rows_on_this_dataset":1,"code_links":14,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":10,"samples_unverified":5,"pointer_only_for_licence":11,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/leveraging-passage-retrieval-with-generative","title":"Leveraging Passage Retrieval with Generative Models for Open Domain Question Answering","date":"2020-07-02","rows_on_this_dataset":1,"code_links":8,"syntology":null},{"paper":"/paper/language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","rows_on_this_dataset":1,"code_links":67,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":65,"samples_ran":15,"samples_unverified":50,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/generating-diverse-and-consistent-qa-pairs","title":"Generating Diverse and Consistent QA pairs from Contexts with Information-Maximizing Hierarchical Conditional VAEs","date":"2020-05-28","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":16,"samples_ran":0,"samples_unverified":16,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/retrieval-augmented-generation-for-knowledge","title":"Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","date":"2020-05-22","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/dense-passage-retrieval-for-open-domain","title":"Dense Passage Retrieval for Open-Domain Question Answering","date":"2020-04-10","rows_on_this_dataset":1,"code_links":19,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":14,"samples_ran":10,"samples_unverified":4,"pointer_only_for_licence":9,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/spanbert-improving-pre-training-by","title":"SpanBERT: Improving Pre-training by Representing and Predicting Spans","date":"2019-07-24","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":15,"samples_ran":3,"samples_unverified":12,"pointer_only_for_licence":4,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/190600300","title":"Latent Retrieval for Weakly Supervised Open Domain Question Answering","date":"2019-06-01","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/memoreader-large-scale-reading-comprehension","title":"MemoReader: Large-Scale Reading Comprehension through Neural Memory Controller","date":"2018-10-01","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/simple-and-effective-multi-paragraph-reading","title":"Simple and Effective Multi-Paragraph Reading Comprehension","date":"2017-10-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":0,"samples_unverified":4,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/memen-multi-layer-embedding-with-memory","title":"MEMEN: Multi-layer Embedding with Memory Networks for Machine Comprehension","date":"2017-07-28","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-integration-of-background-knowledge","title":"Dynamic Integration of Background Knowledge in Neural NLU Systems","date":"2017-06-08","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/reinforced-mnemonic-reader-for-machine","title":"Reinforced Mnemonic Reader for Machine Reading Comprehension","date":"2017-05-08","rows_on_this_dataset":1,"code_links":3,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":23,"samples_harvested":364,"samples_ran":157,"samples_unverified":207,"pointer_only_for_licence":75,"papers_with_no_sample_that_ran":5,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}