{"url":"/task/text-compression","name":"Text Compression","slug":"text-compression","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":43,"papers_with_code":18,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":18,"of":18,"tagged_in_all":43,"items":[{"url":"/paper/llms-may-dominate-information-access-neural","title":"Neural Retrievers are Biased Towards LLM-Generated Content","date":"2023-10-31","arxiv_id":"2310.20501","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/llmzip-lossless-text-compression-using-large","title":"LLMZip: Lossless Text Compression using Large Language Models","date":"2023-06-06","arxiv_id":"2306.04050","repositories_listed":2,"syntology":null},{"url":"/paper/tokalign-efficient-vocabulary-adaptation-via","title":"TokAlign: Efficient Vocabulary Adaptation via Token Alignment","date":"2025-06-04","arxiv_id":"2506.03523","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-multi-document-event-summarization","title":"Scaling Multi-Document Event Summarization: Evaluating Compression vs. Full-Text Approaches","date":"2025-02-10","arxiv_id":"2502.06617","repositories_listed":1,"syntology":null},{"url":"/paper/l3tc-leveraging-rwkv-for-learned-lossless-low","title":"L3TC: Leveraging RWKV for Learned Lossless Low-Complexity Text Compression","date":"2024-12-21","arxiv_id":"2412.16642","repositories_listed":1,"syntology":null},{"url":"/paper/intellectseeker-a-personalized-literature","title":"IntellectSeeker: A Personalized Literature Management System with the Probabilistic Model and Large Language Model","date":"2024-12-10","arxiv_id":"2412.07213","repositories_listed":1,"syntology":null},{"url":"/paper/finezip-pushing-the-limits-of-large-language","title":"FineZip : Pushing the Limits of Large Language Models for Practical Lossless Text Compression","date":"2024-09-25","arxiv_id":"2409.17141","repositories_listed":1,"syntology":null},{"url":"/paper/alphazip-neural-network-enhanced-lossless","title":"AlphaZip: Neural Network-Enhanced Lossless Text Compression","date":"2024-09-23","arxiv_id":"2409.15046","repositories_listed":1,"syntology":null},{"url":"/paper/bpe-gets-picky-efficient-vocabulary","title":"BPE Gets Picky: Efficient Vocabulary Refinement During Tokenizer Training","date":"2024-09-06","arxiv_id":"2409.04599","repositories_listed":1,"syntology":null},{"url":"/paper/xcompress-llm-assisted-python-based-text","title":"XCompress: LLM assisted Python-based text compression toolkit","date":"2024-08-12","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-context-compression-efficiently","title":"Recurrent Context Compression: Efficiently Expanding the Context Window of LLM","date":"2024-06-10","arxiv_id":"2406.06110","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/llmlingua-2-data-distillation-for-efficient","title":"LLMLingua-2: Data Distillation for Efficient and Faithful Task-Agnostic Prompt Compression","date":"2024-03-19","arxiv_id":"2403.12968","repositories_listed":1,"syntology":null},{"url":"/paper/gzip-versus-bag-of-words-for-text","title":"Gzip versus bag-of-words for text classification","date":"2023-07-27","arxiv_id":"2307.15002","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-metric-for-evaluating-semantics","title":"Contextualized Semantic Distance between Highly Overlapped Texts","date":"2021-10-04","arxiv_id":"2110.01176","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-neural-text-compression-with","title":"Data-efficient Neural Text Compression with Interactive Learning","date":"2019-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-batch-noise-contrastive-estimation-approach","title":"A Batch Noise Contrastive Estimation Approach for Training Large Vocabulary Language Models","date":"2017-08-20","arxiv_id":"1708.05997","repositories_listed":1,"syntology":null},{"url":"/paper/authorship-verification-based-on-compression","title":"Authorship Verification based on Compression-Models","date":"2017-06-01","arxiv_id":"1706.00516","repositories_listed":1,"syntology":null},{"url":"/paper/syntactically-informed-text-compression-with","title":"Syntactically Informed Text Compression with Recurrent Neural Networks","date":"2016-08-08","arxiv_id":"1608.02893","repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}