{"url":"/dataset/vulnerability-java-dataset","name":"Vulnerability Java Dataset","full_name":null,"description_markdown":"The dataset consists of two versions: $X_1$ with $P_3$ and $X_1$ without $P_3$, where $P_3$ represents a set of random unchanged functions from vulnerability fixing commits. This dataset is designed for finetuning large language models to detect vulnerabilities in code. It can be used for training and evaluating models in automated vulnerability detection tasks.\r\n\r\nSource: [Finetuning Large Language Models for Vulnerability Detection](https://github.com/rmusab/vul-llm-finetune/tree/main/Datasets)","description_withheld":null,"homepage":"https://github.com/rmusab/vul-llm-finetune/tree/main/Datasets","introduced_date":"2024-03-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/finetuning-large-language-models-for","title":"Finetuning Large Language Models for Vulnerability Detection","first_author":"Alexey Shestov","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Vulnerability Detection","url":"/task/vulnerability-detection","datasets_with_task":"/datasets/task/vulnerability-detection"}],"languages":[],"variants":["Vulnerability Java Dataset"],"data_loaders":[{"repo":"https://github.com/rmusab/vul-llm-finetune","url":"https://github.com/rmusab/vul-llm-finetune","frameworks":[]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/vulnerability-detection-on-vulnerability-java","task":"Vulnerability Detection","dataset_variant":"Vulnerability Java Dataset","rows":2,"metrics":["AUC","F1"],"first_row_in_archive_order":{"model":"WizardCoder","paper":"/paper/finetuning-large-language-models-for","metrics":{"AUC":"0.86","F1":"0.27"},"code_links":[{"title":"rmusab/vul-llm-finetune","url":"https://github.com/rmusab/vul-llm-finetune"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/finetuning-large-language-models-for","title":"Finetuning Large Language Models for Vulnerability Detection","date":"2024-01-30","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}