{"url":"/task/data-compression","name":"Data Compression","slug":"data-compression","description_markdown":null,"categories":[{"name":"Time Series","url":"/area/time-series"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":459,"papers_with_code":131,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/time-series","name":"Time Series Analysis"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":131,"tagged_in_all":459,"items":[{"url":"/paper/xgboost-a-scalable-tree-boosting-system","title":"XGBoost: A Scalable Tree Boosting System","date":"2016-03-09","arxiv_id":"1603.02754","repositories_listed":28,"syntology":{"n":17,"n_ran":0,"n_unverified":17,"n_pointer_only":0}},{"url":"/paper/dnabert-2-efficient-foundation-model-and","title":"DNABERT-2: Efficient Foundation Model and Benchmark For Multi-Species Genome","date":"2023-06-26","arxiv_id":"2306.15006","repositories_listed":6,"syntology":{"n":23,"n_ran":8,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/entropy-law-the-story-behind-data-compression","title":"Entropy Law: The Story Behind Data Compression and LLM Performance","date":"2024-07-09","arxiv_id":"2407.06645","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/an-introduction-to-neural-data-compression","title":"An Introduction to Neural Data Compression","date":"2022-02-14","arxiv_id":"2202.06533","repositories_listed":3,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/transformer-based-transform-coding","title":"Transformer-based Transform Coding","date":"2021-09-29","arxiv_id":null,"repositories_listed":3,"syntology":null},{"url":"/paper/efficient-manifold-and-subspace","title":"Efficient Manifold and Subspace Approximations with Spherelets","date":"2017-06-26","arxiv_id":"1706.08263","repositories_listed":3,"syntology":null},{"url":"/paper/unipcgc-towards-practical-point-cloud","title":"UniPCGC: Towards Practical Point Cloud Geometry Compression via an Efficient Unified Approach","date":"2025-03-24","arxiv_id":"2503.18541","repositories_listed":2,"syntology":null},{"url":"/paper/tokenization-is-more-than-compression","title":"Tokenization Is More Than Compression","date":"2024-02-28","arxiv_id":"2402.18376","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_unverified":3,"n_pointer_only":6}},{"url":"/paper/large-language-model-evaluation-via-matrix","title":"Diff-eRank: A Novel Rank-Based Metric for Evaluating Large Language Models","date":"2024-01-30","arxiv_id":"2401.17139","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/compact-3d-scene-representation-via-self","title":"Compact 3D Scene Representation via Self-Organizing Gaussian Grids","date":"2023-12-19","arxiv_id":"2312.13299","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/bayesflow-amortized-bayesian-workflows-with","title":"BayesFlow: Amortized Bayesian Workflows With Neural Networks","date":"2023-06-28","arxiv_id":"2306.16015","repositories_listed":2,"syntology":null},{"url":"/paper/fast-and-multi-aspect-mining-of-complex-time","title":"Fast and Multi-aspect Mining of Complex Time-stamped Event Streams","date":"2023-03-07","arxiv_id":"2303.03789","repositories_listed":2,"syntology":null},{"url":"/paper/ldmic-learning-based-distributed-multi-view","title":"LDMIC: Learning-based Distributed Multi-view Image Coding","date":"2023-01-24","arxiv_id":"2301.09799","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/qcnn-quadrature-convolutional-neural-network","title":"QuadConv: Quadrature-Based Convolutions with Applications to Non-Uniform PDE Data Compression","date":"2022-11-09","arxiv_id":"2211.05151","repositories_listed":2,"syntology":null},{"url":"/paper/dc-bench-dataset-condensation-benchmark","title":"DC-BENCH: Dataset Condensation Benchmark","date":"2022-07-20","arxiv_id":"2207.09639","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/bottlefit-learning-compressed-representations","title":"BottleFit: Learning Compressed Representations in Deep Neural Networks for Effective and Efficient Split Computing","date":"2022-01-07","arxiv_id":"2201.02693","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-entropy-coding-with-asymmetric","title":"Understanding Entropy Coding With Asymmetric Numeral Systems (ANS): a Statistician's Perspective","date":"2022-01-05","arxiv_id":"2201.01741","repositories_listed":2,"syntology":null},{"url":"/paper/towards-empirical-sandwich-bounds-on-the-rate-1","title":"Towards Empirical Sandwich Bounds on the Rate-Distortion Function","date":"2021-11-23","arxiv_id":"2111.12166","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/supervised-compression-for-resource","title":"Supervised Compression for Resource-Constrained Edge Computing Systems","date":"2021-08-21","arxiv_id":"2108.11898","repositories_listed":2,"syntology":{"n":17,"n_ran":1,"n_unverified":16,"n_pointer_only":1}},{"url":"/paper/redunet-a-white-box-deep-network-from-the","title":"ReduNet: A White-box Deep Network from the Principle of Maximizing Rate Reduction","date":"2021-05-21","arxiv_id":"2105.10446","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/norm-explicit-quantization-improving-vector","title":"Norm-Explicit Quantization: Improving Vector Quantization for Maximum Inner Product Search","date":"2019-11-12","arxiv_id":"1911.04654","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/2506-08306","title":"AstroCompress: A benchmark dataset for multi-purpose compression of astronomical data","date":"2025-06-10","arxiv_id":"2506.08306","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/stationary-mmd-points-for-cubature","title":"Stationary MMD Points for Cubature","date":"2025-05-27","arxiv_id":"2505.20754","repositories_listed":1,"syntology":null},{"url":"/paper/minilongbench-the-low-cost-long-context","title":"MiniLongBench: The Low-cost Long Context Understanding Benchmark for Large Language Models","date":"2025-05-26","arxiv_id":"2505.19959","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-compression-of-neural-networks-and","title":"Efficient compression of neural networks and datasets","date":"2025-05-23","arxiv_id":"2505.17469","repositories_listed":1,"syntology":null},{"url":"/paper/dd-ranking-rethinking-the-evaluation-of","title":"DD-Ranking: Rethinking the Evaluation of Dataset Distillation","date":"2025-05-19","arxiv_id":"2505.13300","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/replaycad-generative-diffusion-replay-for","title":"ReplayCAD: Generative Diffusion Replay for Continual Anomaly Detection","date":"2025-05-10","arxiv_id":"2505.06603","repositories_listed":1,"syntology":null},{"url":"/paper/hybridgs-high-efficiency-gaussian-splatting","title":"HybridGS: High-Efficiency Gaussian Splatting Data Compression using Dual-Channel Sparse Representation and Point Cloud Encoder","date":"2025-05-03","arxiv_id":"2505.01938","repositories_listed":1,"syntology":null},{"url":"/paper/an-extrapolated-and-provably-convergent","title":"An extrapolated and provably convergent algorithm for nonlinear matrix decomposition with the ReLU function","date":"2025-03-31","arxiv_id":"2503.23832","repositories_listed":1,"syntology":null},{"url":"/paper/remote-inference-over-dynamic-links-via","title":"Remote Inference over Dynamic Links via Adaptive Rate Deep Task-Oriented Vector Quantization","date":"2025-01-05","arxiv_id":"2501.02521","repositories_listed":1,"syntology":null}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}