{"url":"/dataset/expository-prose","name":"Expository Prose","full_name":"Expository-Prose-V1","description_markdown":"Expository-Prose-V1 is a collection of specially-curated corpora gathered from diverse sources, ranging from research papers (arXiv) to European Parliament proceedings (EuroParl). It has been specially filtered and curated for the quality of text, depth of reasoning and breadth of knowledge to faciliate an effective pre-train. It was used to pre-train 1.5-Pints, a small but powerful Large Language Model developed by the Pints Research Team.","description_withheld":null,"homepage":"https://huggingface.co/datasets/pints-ai/Expository-Prose-V1","introduced_date":"2024-08-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/1-5-pints-technical-report-pretraining-in","title":"1.5-Pints Technical Report: Pretraining in Days, Not Months -- Your Language Model Thrives on Quality Data","first_author":"Calvin Tan","url":null},"license":{"name":"MIT","url":"https://choosealicense.com/licenses/mit/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Text Generation","url":"/task/text-generation","datasets_with_task":"/datasets/task/text-generation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["Expository Prose"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}