{"url":"/task/dataset-distillation","name":"Dataset Distillation","slug":"dataset-distillation","description_markdown":"Dataset distillation is the task of synthesizing a small dataset such that models trained on it achieve high performance on the original large dataset. A dataset distillation algorithm takes as input a large real dataset to be distilled (training set), and outputs a small synthetic distilled dataset, which is evaluated via testing models trained on this distilled dataset on a separate real dataset (validation/test set). A good small distilled dataset is not only useful in dataset understanding, but has various applications (e.g., continual learning, privacy, neural architecture search, etc.).","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":216,"papers_with_code":120,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":120,"tagged_in_all":216,"items":[{"url":"/paper/dataset-distillation-by-matching-training","title":"Dataset Distillation by Matching Training Trajectories","date":"2022-03-22","arxiv_id":"2203.11932","repositories_listed":6,"syntology":null},{"url":"/paper/dataset-distillation","title":"Dataset Distillation","date":"2018-11-27","arxiv_id":"1811.10959","repositories_listed":6,"syntology":{"n":17,"n_ran":3,"n_unverified":14,"n_pointer_only":3}},{"url":"/paper/on-the-diversity-and-realism-of-distilled","title":"On the Diversity and Realism of Distilled Dataset: An Efficient Dataset Distillation Paradigm","date":"2023-12-06","arxiv_id":"2312.03526","repositories_listed":4,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":4}},{"url":"/paper/fedcache-2-0-exploiting-the-potential-of","title":"FedCache 2.0: Federated Edge Learning with Knowledge Caching and Dataset Distillation","date":"2024-05-22","arxiv_id":"2405.13378","repositories_listed":3,"syntology":null},{"url":"/paper/minimizing-the-accumulated-trajectory-error","title":"Minimizing the Accumulated Trajectory Error to Improve Dataset Distillation","date":"2022-11-20","arxiv_id":"2211.11004","repositories_listed":3,"syntology":null},{"url":"/paper/dataset-distillation-via-factorization","title":"Dataset Distillation via Factorization","date":"2022-10-30","arxiv_id":"2210.16774","repositories_listed":3,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/dataset-distillation-with-infinitely-wide","title":"Dataset Distillation with Infinitely Wide Convolutional Networks","date":"2021-07-27","arxiv_id":"2107.13034","repositories_listed":3,"syntology":null},{"url":"/paper/improving-dataset-distillation","title":"Soft-Label Dataset Distillation and Text Dataset Distillation","date":"2019-10-06","arxiv_id":"1910.02551","repositories_listed":3,"syntology":{"n":13,"n_ran":0,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/dataset-distillation-via-vision-language","title":"Dataset Distillation via Vision-Language Category Prototype","date":"2025-06-30","arxiv_id":"2506.23580","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/generative-dataset-distillation-based-on","title":"Generative Dataset Distillation Based on Diffusion Model","date":"2024-08-16","arxiv_id":"2408.08610","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/embarassingly-simple-dataset-distillation","title":"Embarassingly Simple Dataset Distillation","date":"2023-11-13","arxiv_id":"2311.07025","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/does-graph-distillation-see-like-vision","title":"Does Graph Distillation See Like Vision Dataset Counterpart?","date":"2023-10-13","arxiv_id":"2310.09192","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/self-supervised-set-representation-learning","title":"Self-Supervised Dataset Distillation for Transfer Learning","date":"2023-10-10","arxiv_id":"2310.06511","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/datadam-efficient-dataset-distillation-with-1","title":"DataDAM: Efficient Dataset Distillation with Attention Matching","date":"2023-09-29","arxiv_id":"2310.00093","repositories_listed":2,"syntology":{"n":43,"n_ran":35,"n_unverified":8,"n_pointer_only":43}},{"url":"/paper/multimodal-dataset-distillation-for-image","title":"Vision-Language Dataset Distillation","date":"2023-08-15","arxiv_id":"2308.07545","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/towards-trustworthy-dataset-distillation","title":"Towards Trustworthy Dataset Distillation","date":"2023-07-18","arxiv_id":"2307.09165","repositories_listed":2,"syntology":null},{"url":"/paper/squeeze-recover-and-relabel-dataset","title":"Squeeze, Recover and Relabel: Dataset Condensation at ImageNet Scale From A New Perspective","date":"2023-06-22","arxiv_id":"2306.13092","repositories_listed":2,"syntology":{"n":6,"n_ran":1,"n_unverified":5,"n_pointer_only":6}},{"url":"/paper/distill-gold-from-massive-ores-efficient","title":"Distill Gold from Massive Ores: Bi-level Data Pruning towards Efficient Dataset Distillation","date":"2023-05-28","arxiv_id":"2305.18381","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/generalizing-dataset-distillation-via-deep","title":"Generalizing Dataset Distillation via Deep Generative Prior","date":"2023-05-02","arxiv_id":"2305.01649","repositories_listed":2,"syntology":null},{"url":"/paper/dim-distilling-dataset-into-generative-model","title":"DiM: Distilling Dataset into Generative Model","date":"2023-03-08","arxiv_id":"2303.04707","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/dream-efficient-dataset-distillation-by","title":"DREAM: Efficient Dataset Distillation by Representative Matching","date":"2023-02-28","arxiv_id":"2302.14416","repositories_listed":2,"syntology":null},{"url":"/paper/dataset-distillation-with-convexified","title":"Dataset Distillation with Convexified Implicit Gradients","date":"2023-02-13","arxiv_id":"2302.06755","repositories_listed":2,"syntology":null},{"url":"/paper/backdoor-attacks-against-dataset-distillation","title":"Backdoor Attacks Against Dataset Distillation","date":"2023-01-03","arxiv_id":"2301.01197","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/accelerating-dataset-distillation-via-model","title":"Accelerating Dataset Distillation via Model Augmentation","date":"2022-12-12","arxiv_id":"2212.06152","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/scaling-up-dataset-distillation-to-imagenet","title":"Scaling Up Dataset Distillation to ImageNet-1K with Constant Memory","date":"2022-11-19","arxiv_id":"2211.10586","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/efficient-dataset-distillation-using-random","title":"Efficient Dataset Distillation Using Random Feature Approximation","date":"2022-10-21","arxiv_id":"2210.12067","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/federated-learning-via-decentralized-dataset","title":"Federated Learning via Decentralized Dataset Distillation in Resource-Constrained Edge Environments","date":"2022-08-24","arxiv_id":"2208.11311","repositories_listed":2,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/remember-the-past-distilling-datasets-into","title":"Remember the Past: Distilling Datasets into Addressable Memories for Neural Networks","date":"2022-06-06","arxiv_id":"2206.02916","repositories_listed":2,"syntology":null},{"url":"/paper/dataset-distillation-using-neural-feature","title":"Dataset Distillation using Neural Feature Regression","date":"2022-06-01","arxiv_id":"2206.00719","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/flexible-dataset-distillation-learn-labels","title":"Flexible Dataset Distillation: Learn Labels Instead of Images","date":"2020-06-15","arxiv_id":"2006.08572","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_unverified":2,"n_pointer_only":0}}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}