{"url":"/dataset/vulscriber","name":"VulScribeR","full_name":"VulScriber: 22K+ unfiltered vul samples generated with ChatGPT via Injection","description_markdown":"Datasets are listed in the repository's readme file. This one is extra and yields 20K+ items after filtering with a fuzzy parser.","description_withheld":null,"homepage":"","introduced_date":"2024-08-07","introduced_date_note":null,"introduced_by":{"paper":"/paper/exploring-rag-based-vulnerability","title":"VulScribeR: Exploring RAG-based Vulnerability Augmentation with LLMs","first_author":"Seyed Shayan Daneshvar","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Vulnerability Detection","url":"/task/vulnerability-detection","datasets_with_task":"/datasets/task/vulnerability-detection"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["VulScribeR"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/vulnerability-detection-on-vulscriber","task":"Vulnerability Detection","dataset_variant":"VulScribeR","rows":6,"metrics":["F1 Score"],"first_row_in_archive_order":{"model":"Reveal Model - Tested on Reveal (Training on Devign + VulScribeR 20K + Extra Cleans)","paper":"/paper/exploring-rag-based-vulnerability","metrics":{"F1 Score":"26.18"},"code_links":[{"title":"VulScribeR/VulScribeR","url":"https://github.com/VulScribeR/VulScribeR"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/exploring-rag-based-vulnerability","title":"VulScribeR: Exploring RAG-based Vulnerability Augmentation with LLMs","date":"2024-08-07","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}