{"url":"/dataset/fulg","name":"FuLG","full_name":null,"description_markdown":"FuLG is a comprehensive Romanian language corpus comprising 150 billion tokens, carefully extracted from Common Crawl. This extensive dataset is the result of rigorous filtering and deduplication processes applied to 95 Common Crawl snapshots. The compressed dataset has 289 GB.","description_withheld":null,"homepage":"https://huggingface.co/datasets/faur-ai/fulg","introduced_date":"2024-07-18","introduced_date_note":null,"introduced_by":{"paper":"/paper/fulg-150b-romanian-corpus-for-language-model","title":"FuLG: 150B Romanian Corpus for Language Model Pretraining","first_author":"Vlad-Andrei Bădoiu","url":null},"license":{"name":"ODC-BY","url":"https://huggingface.co/datasets/faur-ai/fulg/blob/main/LICENSE.md"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[],"languages":[{"name":"Romanian","url":"/datasets/language/romanian"}],"variants":["FuLG"],"data_loaders":[],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}