{"url":"/dataset/curlie","name":"Curlie","full_name":null,"description_markdown":"**Curlie dataset** is a dataset with more than 1M websites in 92 languages with relative labels collected from Curlie, the largest multilingual crowdsourced Web directory. The dataset contains 14 website categories aligned across languages. It is used for language-agnostic website embedding and classification","description_withheld":null,"homepage":"https://doi.org/10.6084/m9.figshare.16621669","introduced_date":"2022-01-10","introduced_date_note":null,"introduced_by":{"paper":"/paper/language-agnostic-website-embedding-and","title":"Homepage2Vec: Language-Agnostic Website Embedding and Classification","first_author":"Sylvain Lugeon","url":null},"license":null,"modalities":[],"tasks":[],"languages":[],"variants":["Curlie"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}