{"url":"/dataset/websrc","name":"WebSRC","full_name":"WebSRC: A Dataset for Web-Based Structural Reading Comprehension","description_markdown":"WebSRC is a novel Web-based Structural Reading Comprehension dataset. It consists of 0.44M question-answer pairs, which are collected from 6.5K web pages with corresponding HTML source code, screenshots and metadata. Each question in WebSRC requires a certain structural understanding of a web page to answer, and the answer is either a text span on the web page or yes/no.\r\n\r\nSource: [WebSRC Homepage](https://x-lance.github.io/WebSRC/)","description_withheld":null,"homepage":"https://x-lance.github.io/WebSRC/","introduced_date":"2021-01-23","introduced_date_note":null,"introduced_by":{"paper":"/paper/websrc-a-dataset-for-web-based-structural","title":"WebSRC: A Dataset for Web-Based Structural Reading Comprehension","first_author":"Xingyu Chen","url":null},"license":{"name":"MIT License","url":"https://github.com/X-LANCE/WebSRC-Baseline?tab=MIT-1-ov-file#readme"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Tables","url":"/datasets/modality/tables"}],"tasks":[{"name":"Question Answering","url":"/task/question-answering","datasets_with_task":"/datasets/task/question-answering"},{"name":"Visual Question Answering (VQA)","url":"/task/visual-question-answering","datasets_with_task":"/datasets/task/visual-question-answering"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["WebSRC"],"data_loaders":[],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/question-answering-on-websrc","task":"Question Answering","dataset_variant":"WebSRC","rows":1,"metrics":["F1"],"first_row_in_archive_order":{"model":"ChatGPT 3.5 SpatialFormat","paper":"/paper/lapdoc-layout-aware-prompting-for-documents","metrics":{"F1":"80.7"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-question-answering-vqa-on-websrc","task":"Visual Question Answering (VQA)","dataset_variant":"WebSRC","rows":1,"metrics":["EM"],"first_row_in_archive_order":{"model":"DUBLIN","paper":"/paper/dublin-document-understanding-by-language","metrics":{"EM":"77.75"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/lapdoc-layout-aware-prompting-for-documents","title":"LAPDoc: Layout-Aware Prompting for Documents","date":"2024-02-15","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dublin-document-understanding-by-language","title":"DUBLIN -- Document Understanding By Language-Image Network","date":"2023-05-23","rows_on_this_dataset":1,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}