{"url":"/dataset/pubmedabstractssubsetembedded","name":"PubMedAbstractsSubsetEmbedded","full_name":null,"description_markdown":"This dataset contains a probabilistic sample of ~2.4 million PubMed abstracts, enriched with precomputed dense embeddings (title + abstract), from the **`ncbi/MedCPT-Article-Encoder`** model. It is derived from public metadata made available via the [National Library of Medicine (NLM)](https://pubmed.ncbi.nlm.nih.gov/) and was used in the paper [*Efficient and Reproducible Biomedical QA using Retrieval-Augmented Generation*](https://arxiv.org/abs/2505.07917).\r\n\r\nEach entry includes:\r\n- `title`: Title of the publication\r\n- `abstract`: Abstract content\r\n- `PMID`: PubMed identifier\r\n- `embedding`: 768-dimensional float32 vector from MedCPT","description_withheld":null,"homepage":"https://huggingface.co/datasets/slinusc/PubMedAbstractsSubsetEmbedded","introduced_date":"2025-03-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/efficient-and-reproducible-biomedical","title":"Efficient and Reproducible Biomedical Question Answering using Retrieval Augmented Generation","first_author":"Linus Stuhlmann","url":null},"license":{"name":"CC-BY-4.0","url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/cc-by-4.0.md"},"modalities":[],"tasks":[],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["PubMedAbstractsSubsetEmbedded"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}