{"url":"/dataset/musk-v1","name":"Musk v1","full_name":"Musk v1","description_markdown":"The Musk dataset describes a set of molecules, and the objective is to detect musks from non-musks. This dataset describes a set of 92 molecules of which 47 are judged by human experts to be musks and the remaining 45 molecules are judged to be non-musks. There are 166 features available that describe the molecules based on the shape of the molecule.\n\nSource: [Estimation of Dimensions Contributing to Detected Anomalies with Variational Autoencoders](https://arxiv.org/abs/1811.04576)","description_withheld":null,"homepage":"https://archive.ics.uci.edu/ml/datasets/Musk+(Version+1)","introduced_date":"1997-01-01","introduced_date_note":null,"introduced_by":{"paper":"/paper/solving-the-multiple-instance-problem-with","title":"Solving the multiple instance problem with axis-parallel rectangles","first_author":"Thomas G. Dietteric","url":null},"license":null,"modalities":[{"name":"Tabular","url":"/datasets/modality/tabular"}],"tasks":[{"name":"Anomaly Detection","url":"/task/anomaly-detection","datasets_with_task":"/datasets/task/anomaly-detection"},{"name":"Multiple Instance Learning","url":"/task/multiple-instance-learning","datasets_with_task":"/datasets/task/multiple-instance-learning"}],"languages":[],"variants":["Musk v1"],"data_loaders":[],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multiple-instance-learning-on-musk-v1","task":"Multiple Instance Learning","dataset_variant":"Musk v1","rows":2,"metrics":["ACC","AUC"],"first_row_in_archive_order":{"model":"Snuffy","paper":"/paper/snuffy-efficient-whole-slide-image-classifier","metrics":{"ACC":"0.961","AUC":"0.989"},"code_links":[{"title":"jafarinia/snuffy","url":"https://github.com/jafarinia/snuffy"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/anomaly-detection-on-musk-v1","task":"Anomaly Detection","dataset_variant":"Musk v1","rows":1,"metrics":["F1-Score"],"first_row_in_archive_order":{"model":"RCALAD","paper":"/paper/regularized-complete-cycle-consistent-gan-for","metrics":{"F1-Score":"63.1"},"code_links":[{"title":"zahradehghanian97/rcalad","url":"https://github.com/zahradehghanian97/rcalad"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/snuffy-efficient-whole-slide-image-classifier","title":"Snuffy: Efficient Whole Slide Image Classifier","date":"2024-08-15","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/regularized-complete-cycle-consistent-gan-for","title":"Spot The Odd One Out: Regularized Complete Cycle Consistent Anomaly Detector GAN","date":"2023-04-16","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dual-stream-multiple-instance-learning","title":"Dual-stream Multiple Instance Learning Network for Whole Slide Image Classification with Self-supervised Contrastive Learning","date":"2020-11-17","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":6,"samples_ran":4,"samples_unverified":2,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}