{"url":"/dataset/dataset-of-legal-documents","name":"Dataset of Legal Documents","full_name":null,"description_markdown":"Dataset of Legal Documents consists of court decisions from 2017 and 2018 were selected for the dataset, published online by the Federal Ministry of Justice and Consumer Protection. The documents originate from seven federal courts: Federal Labour Court (BAG), Federal Fiscal Court (BFH), Federal Court of Justice (BGH), Federal Patent Court (BPatG), Federal Social Court (BSG), Federal Constitutional Court (BVerfG) and Federal Administrative Court (BVerwG).\r\n\r\nThe dataset consists of 66,723 sentences with 2,157,048 tokens. The sizes of the seven court-specific datasets varies between 5,858 and 12,791 sentences, and 177,835 to 404,041 tokens. The distribution of annotations on a per-token basis corresponds to approx. 19-23 %.\r\n\r\nSource: [GitHub](https://github.com/elenanereiss/Legal-Entity-Recognition)","description_withheld":null,"homepage":"https://github.com/elenanereiss/Legal-Entity-Recognition","introduced_date":"2020-03-29","introduced_date_note":null,"introduced_by":{"paper":"/paper/a-dataset-of-german-legal-documents-for-named","title":"A Dataset of German Legal Documents for Named Entity Recognition","first_author":"Elena Leitner","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"}],"languages":[{"name":"German","url":"/datasets/language/german"}],"variants":["Dataset of Legal Documents","elenanereiss/german-ler"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/elenanereiss/german-ler","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/elenanereiss/Legal-Entity-Recognition","url":"https://github.com/elenanereiss/Legal-Entity-Recognition","frameworks":[]}],"num_papers_in_archive":2,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}