{"url":"/task/text-segmentation","name":"Text Segmentation","slug":"text-segmentation","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":124,"papers_with_code":40,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/conll-1","name":"CoNLL","full_name":"","num_papers_in_archive":187},{"url":"/dataset/spmrl-hebrew-segmentation-data","name":"SPMRL Hebrew segmentation data","full_name":"","num_papers_in_archive":2},{"url":"/dataset/ytseg","name":"YTSeg","full_name":"","num_papers_in_archive":2},{"url":"/dataset/careercoach-2022","name":"CareerCoach 2022","full_name":"","num_papers_in_archive":1},{"url":"/dataset/conll-2017-shared-task-automatically","name":"CoNLL 2017 Shared Task - Automatically Annotated Raw Texts and Word Embeddings","full_name":"","num_papers_in_archive":1},{"url":"/dataset/urdudoc","name":"UrduDoc","full_name":"","num_papers_in_archive":1},{"url":"/dataset/wiki5k-hebrew-segmentation","name":"Wiki5K Hebrew segmentation","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":40,"tagged_in_all":124,"items":[{"url":"/paper/cotype-joint-extraction-of-typed-entities-and","title":"CoType: Joint Extraction of Typed Entities and Relations with Knowledge Bases","date":"2016-10-27","arxiv_id":"1610.08763","repositories_listed":3,"syntology":null},{"url":"/paper/segmenting-text-and-learning-their-rewards","title":"Segmenting Text and Learning Their Rewards for Improved RLHF in Language Model","date":"2025-01-06","arxiv_id":"2501.02790","repositories_listed":2,"syntology":null},{"url":"/paper/wenetspeech-a-10000-hours-multi-domain","title":"WenetSpeech: A 10000+ Hours Multi-domain Mandarin Corpus for Speech Recognition","date":"2021-10-07","arxiv_id":"2110.03370","repositories_listed":2,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/text-segmentation-as-a-supervised-learning","title":"Text Segmentation as a Supervised Learning Task","date":"2018-03-25","arxiv_id":"1803.09337","repositories_listed":2,"syntology":null},{"url":"/paper/sequence-modeling-via-segmentations","title":"Sequence Modeling via Segmentations","date":"2017-02-24","arxiv_id":"1702.07463","repositories_listed":2,"syntology":null},{"url":"/paper/controltext-unlocking-controllable-fonts-in","title":"ControlText: Unlocking Controllable Fonts in Multilingual Text Rendering without Font Annotations","date":"2025-02-16","arxiv_id":"2502.10999","repositories_listed":1,"syntology":null},{"url":"/paper/meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","arxiv_id":"2410.12788","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/was-dataset-and-methods-for-artistic-text","title":"WAS: Dataset and Methods for Artistic Text Segmentation","date":"2024-07-31","arxiv_id":"2408.00106","repositories_listed":1,"syntology":null},{"url":"/paper/eaformer-scene-text-segmentation-with-edge","title":"EAFormer: Scene Text Segmentation with Edge-Aware Transformers","date":"2024-07-24","arxiv_id":"2407.17020","repositories_listed":1,"syntology":null},{"url":"/paper/towards-detecting-ai-generated-text-within","title":"Detecting AI-Generated Sentences in Human-AI Collaborative Hybrid Texts: Challenges, Strategies, and Insights","date":"2024-03-06","arxiv_id":"2403.03506","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":6}},{"url":"/paper/hi-sam-marrying-segment-anything-model-for","title":"Hi-SAM: Marrying Segment Anything Model for Hierarchical Text Segmentation","date":"2024-01-31","arxiv_id":"2401.17904","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/filtered-semi-markov-crf","title":"Filtered Semi-Markov CRF","date":"2023-11-29","arxiv_id":"2311.18028","repositories_listed":1,"syntology":null},{"url":"/paper/psstrnet-progressive-segmentation-guided","title":"PSSTRNet: Progressive Segmentation-guided Scene Text Removal Network","date":"2023-06-13","arxiv_id":"2306.07842","repositories_listed":1,"syntology":null},{"url":"/paper/expanding-scope-adapting-english-adversarial","title":"Expanding Scope: Adapting English Adversarial Attacks to Chinese","date":"2023-06-08","arxiv_id":"2306.04874","repositories_listed":1,"syntology":null},{"url":"/paper/ccdwt-gan-generative-adversarial-networks","title":"CCDWT-GAN: Generative Adversarial Networks Based on Color Channel Using Discrete Wavelet Transform for Document Image Binarization","date":"2023-05-27","arxiv_id":"2305.17420","repositories_listed":1,"syntology":null},{"url":"/paper/three-stage-binarization-of-color-document","title":"Three-stage binarization of color document images based on discrete wavelet transform and generative adversarial networks","date":"2022-11-29","arxiv_id":"2211.16098","repositories_listed":1,"syntology":null},{"url":"/paper/dq-detr-dual-query-detection-transformer-for","title":"DQ-DETR: Dual Query Detection Transformer for Phrase Extraction and Grounding","date":"2022-11-28","arxiv_id":"2211.15516","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-character-to-character","title":"Self-supervised Character-to-Character Distillation for Text Recognition","date":"2022-11-01","arxiv_id":"2211.00288","repositories_listed":1,"syntology":null},{"url":"/paper/toward-unifying-text-segmentation-and-long","title":"Toward Unifying Text Segmentation and Long Document Summarization","date":"2022-10-28","arxiv_id":"2210.16422","repositories_listed":1,"syntology":null},{"url":"/paper/a-glyph-driven-topology-enhancement-network","title":"Self-supervised Implicit Glyph Attention for Text Recognition","date":"2022-03-07","arxiv_id":"2203.03382","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-discourse-segmentation-for","title":"Weakly supervised discourse segmentation for multiparty oral conversations","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sefamerve-arge-at-semeval-2021-task-5-toxic","title":"Sefamerve ARGE at SemEval-2021 Task 5: Toxic Spans Detection Using Segmentation Based 1-D Convolutional Neural Network Model","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sequence-segmentation-as-determining","title":"Neural Sequence Segmentation as Determining the Leftmost Segments","date":"2021-04-15","arxiv_id":"2104.07217","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":8}},{"url":"/paper/topical-change-detection-in-documents-via","title":"Structural Text Segmentation of Legal Documents","date":"2020-12-07","arxiv_id":"2012.03619","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-text-segmentation-for-medieval","title":"Hierarchical Text Segmentation for Medieval Manuscripts","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-text-segmentation-a-novel-dataset","title":"Rethinking Text Segmentation: A Novel Dataset and A Text-Specific Refinement Approach","date":"2020-11-27","arxiv_id":"2011.14021","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-natural-language-segmentation","title":"Interpretable Natural Language Segmentation Based on Link Grammar","date":"2020-11-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/chapter-captor-text-segmentation-in-novels","title":"Chapter Captor: Text Segmentation in Novels","date":"2020-11-09","arxiv_id":"2011.04163","repositories_listed":1,"syntology":null},{"url":"/paper/improving-segmentation-for-technical-support","title":"Improving Segmentation for Technical Support Problems","date":"2020-05-22","arxiv_id":"2005.11055","repositories_listed":1,"syntology":null},{"url":"/paper/text-segmentation-by-cross-segment-attention","title":"Text Segmentation by Cross Segment Attention","date":"2020-04-30","arxiv_id":"2004.14535","repositories_listed":1,"syntology":null}],"syntology_records":5,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}