{"url":"/dataset/vcr-wiki","name":"VCR-wiki","full_name":"VCR-wiki","description_markdown":"This task stems from the observation that text embedded in images is intrinsically different from common visual elements and natural language due to the need to align the modalities of vision, text, and text embedded in images. \r\n\r\nSolving OCR usually does not depend on the context, while VQA usually does not have a unique solution to evaluate the quality of answers. VCR bridges the gap between OCR and VQA: it reconstructs the unique text in images while considering the context of the rest image.","description_withheld":null,"homepage":"https://github.com/tianyu-z/vcr","introduced_date":"2024-06-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","first_author":"Tianyu Zhang","url":null},"license":{"name":"CC-BY-SA","url":"https://github.com/tianyu-z/VCR/blob/main/LICENSE-CC-BY-SA"},"modalities":[{"name":"Images","url":"/datasets/modality/images"},{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Chinese","url":"/datasets/language/chinese"}],"variants":["VCR-wiki"],"data_loaders":[{"repo":"https://github.com/tianyu-z/vcr","url":"https://huggingface.co/vcr-org","frameworks":["pytorch"]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}