{"url":"/dataset/paws","name":"PAWS","full_name":"Paraphrase Adversaries from Word Scrambling","description_markdown":"Paraphrase Adversaries from Word Scrambling (PAWS) is a dataset contains 108,463 human-labeled and 656k noisily labeled pairs that feature the importance of modeling structure, context, and word order information for the problem of paraphrase identification. The dataset has two subsets, one based on Wikipedia and the other one based on the Quora Question Pairs (QQP) dataset.\r\n\r\nSource: [PAWS](https://github.com/google-research-datasets/paws)","description_withheld":null,"homepage":"https://github.com/google-research-datasets/paws","introduced_date":"2019-04-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/paws-paraphrase-adversaries-from-word","title":"PAWS: Paraphrase Adversaries from Word Scrambling","first_author":"Yuan Zhang","url":null},"license":{"name":"Custom","url":"https://github.com/google-research-datasets/paws"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Natural Language Inference","url":"/task/natural-language-inference","datasets_with_task":"/datasets/task/natural-language-inference"},{"name":"Paraphrase Identification","url":"/task/paraphrase-identification","datasets_with_task":"/datasets/task/paraphrase-identification"},{"name":"Data Augmentation","url":"/task/data-augmentation","datasets_with_task":"/datasets/task/data-augmentation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["PAWS"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/google-research-datasets/paws","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/paws","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/paws_wiki","frameworks":["tf","jax"]},{"repo":"https://github.com/google-research-datasets/paws","url":"https://github.com/google-research-datasets/paws","frameworks":[]}],"num_papers_in_archive":159,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}