{"url":"/dataset/pathbased","name":"pathbased","full_name":null,"description_markdown":"**pathbased** is a 3-cluster data set. The data set consists of a circular cluster with an opening near the bottom and two Gaussian distributed clusters inside. Each cluster contains 100 data points.\r\n\r\nSource: [Robust path-based spectral clustering](https://www.sciencedirect.com/science/article/abs/pii/S0031320307002038)","description_withheld":null,"homepage":"http://cs.uef.fi/sipu/datasets/pathbased.txt","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/robust-path-based-spectral-clustering","title":"Robust path-based spectral clustering","first_author":"Hong Chang","url":null},"license":{"name":"Unknown","url":null},"modalities":[],"tasks":[{"name":"Clustering Algorithms Evaluation","url":"/task/clustering-algorithms-evaluation","datasets_with_task":"/datasets/task/clustering-algorithms-evaluation"},{"name":"Outlier Detection","url":"/task/outlier-detection","datasets_with_task":"/datasets/task/outlier-detection"},{"name":"Image/Document Clustering","url":"/task/imagedocument-clustering","datasets_with_task":"/datasets/task/imagedocument-clustering"},{"name":"Clustering Ensemble","url":"/task/clustering-ensemble","datasets_with_task":"/datasets/task/clustering-ensemble"}],"languages":[],"variants":["pathbased"],"data_loaders":[],"num_papers_in_archive":25,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/clustering-algorithms-evaluation-on-pathbased","task":"Clustering Algorithms Evaluation","dataset_variant":"pathbased","rows":1,"metrics":["Purity"],"first_row_in_archive_order":{"model":"CVDD","paper":"/paper/an-internal-validity-index-based-on-density","metrics":{"Purity":"0.977"},"code_links":[{"title":"hulianyu/CVDD","url":"https://github.com/hulianyu/CVDD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/clustering-ensemble-on-pathbased","task":"Clustering Ensemble","dataset_variant":"pathbased","rows":1,"metrics":["Purity"],"first_row_in_archive_order":{"model":"NegMM","paper":"/paper/ensemble-clustering-based-on-evidence","metrics":{"Purity":"0.98"},"code_links":[{"title":"zhongcaiming/clustering-ensemble","url":"https://github.com/zhongcaiming/clustering-ensemble"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/ensemble-clustering-based-on-evidence","title":"Ensemble clustering based on evidence extracted from the co-association matrix","date":"2019-08-01","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/an-internal-validity-index-based-on-density","title":"An Internal Validity Index Based on Density-Involved Distance","date":"2019-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}