{"url":"/dataset/google-ranked-urls-dataset","name":"Google Ranked URLs Dataset","full_name":null,"description_markdown":"This dataset was curated for Search Engine Optimization (SEO) analysis tasks, including categorization and spam detection. It covers 12 diverse topics: basketball, books, cats, gardening, history, movies, music, recipes, sports, technology, travel, and weather. Some topics have hierarchical relationships, such as sports and basketball, while others are closely related (e.g., movies and music) or unrelated (e.g., basketball and gardening), with varying degrees of overlap among them. For each topic, approximately 300 search queries were generated using large language models (LLMs) like GPT, Llama, and Claude. The top 10 URLs from the Google Search Console’s search engine results page (SERP) were retrieved for each query.","description_withheld":null,"homepage":"https://github.com/FardinRastakhiz/QuickCharNet/tree/main/datasets/extractedURLs","introduced_date":"2024-10-22","introduced_date_note":null,"introduced_by":{"paper":"/paper/quickcharnet-an-efficient-url-classification","title":"QuickCharNet: An Efficient URL Classification Framework for Enhanced Search Engine Optimization","first_author":"Fardin Rastakhiz","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Classification","url":"/task/classification-1","datasets_with_task":"/datasets/task/classification-1"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Google Ranked URLs Dataset"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}