{"url":"/dataset/cath-4-3","name":"CATH 4.3","full_name":null,"description_markdown":"The CATH (Class, Architecture, Topology, Homology) [65] database is a comprehensive\r\nresource for protein structure classification that hierarchical group proteins based on their structural\r\nfeatures. The database defines classes based on topological similarities, architectures based on the\r\narrangement of secondary structure elements, topologies based on the connectivity of secondary\r\nstructure elements, and homologous domains based on sequence similarity.  This results in a training set of 16,153 structures, a validation set of\r\n1,457 structures, and a test set of 1,797 structures. Note that the curated CATH dataset contains only\r\nsingle-chain structures and does not consider the case of designing multi-chain proteins.","description_withheld":null,"homepage":"https://cathdb.info/wiki/doku/?id=index","introduced_date":null,"introduced_date_note":null,"introduced_by":null,"license":null,"modalities":[],"tasks":[{"name":"Protein Design","url":"/task/protein-design","datasets_with_task":"/datasets/task/protein-design"},{"name":"Protein Language Model","url":"/task/protein-language-model","datasets_with_task":"/datasets/task/protein-language-model"}],"languages":[],"variants":["CATH 4.3"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/protein-design-on-cath-4-3","task":"Protein Design","dataset_variant":"CATH 4.3","rows":2,"metrics":["Perplexity","Sequence Recovery %(All)"],"first_row_in_archive_order":{"model":"GVP-large","paper":"/paper/knowledge-design-pushing-the-limit-of-protein","metrics":{"Perplexity":"6.17","Sequence Recovery %(All)":"39.2"},"code_links":[{"title":"A4Bio/OpenCPD","url":"https://github.com/A4Bio/OpenCPD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/knowledge-design-pushing-the-limit-of-protein","title":"Knowledge-Design: Pushing the Limit of Protein Design via Knowledge Refinement","date":"2023-05-20","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}