{"url":"/task/chunking","name":"Chunking","slug":"chunking","description_markdown":"Chunking, also known as shallow parsing, identifies continuous spans of tokens that form syntactic units such as noun phrases or verb phrases.\r\n\r\nExample:\r\n\r\n| Vinken | , | 61 | years | old |\r\n| --- | ---| --- | --- | --- |\r\n| B-NLP| I-NP | I-NP | I-NP | I-NP |","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":447,"papers_with_code":120,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":5,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/chunking-on-conll-2000","slug":"chunking-on-conll-2000","dataset":"CoNLL 2000","dataset_url":"/dataset/conll-1","rows_in_archive":9,"metrics":["Exact Span F1"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/chunking-on-penn-treebank","slug":"chunking-on-penn-treebank","dataset":"Penn Treebank","dataset_url":"/dataset/penn-treebank","rows_in_archive":8,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/chunking-on-conll-2003-english","slug":"chunking-on-conll-2003-english","dataset":"CoNLL 2003 (English)","dataset_url":"/dataset/conll-1","rows_in_archive":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/chunking-on-conll-2003-german","slug":"chunking-on-conll-2003-german","dataset":"CoNLL 2003 (German)","dataset_url":"/dataset/conll-1","rows_in_archive":3,"metrics":["F1"],"first_row_in_archive_order":{"model":"ACE","paper_title":"Automated Concatenation of Embeddings for Structured Prediction","paper_url":"/paper/automated-concatenation-of-embeddings-for-1","paper_date":"2020-10-10","arxiv_id":"2010.05006","code_links":[{"title":"Alibaba-NLP/ACE","url":"https://github.com/Alibaba-NLP/ACE"},{"title":"zhaoyuesun/phee","url":"https://github.com/zhaoyuesun/phee"}],"syntology":null}},{"leaderboard":"/sota/chunking-on-conll-2003","slug":"chunking-on-conll-2003","dataset":"CoNLL 2003","dataset_url":"/dataset/conll-2003","rows_in_archive":1,"metrics":["AUC","Accuracy","F1","Precision","Recall"],"first_row_in_archive_order":{"model":"Def2Vec","paper_title":"Def2Vec: Extensible Word Embeddings from Dictionary Definitions","paper_url":"/paper/def2vec-extensible-word-embeddings-from","paper_date":"2023-12-16","arxiv_id":null,"code_links":[{"title":"IreneMorazzoni/def_2_vec_irene","url":"https://github.com/IreneMorazzoni/def_2_vec_irene"},{"title":"vincenzo-scotti/def_2_vec","url":"https://github.com/vincenzo-scotti/def_2_vec"}],"syntology":null}}],"datasets":[{"url":"/dataset/penn-treebank","name":"Penn Treebank","full_name":"","num_papers_in_archive":1006},{"url":"/dataset/conll-2003","name":"CoNLL 2003","full_name":"","num_papers_in_archive":755},{"url":"/dataset/conll-1","name":"CoNLL","full_name":"","num_papers_in_archive":187},{"url":"/dataset/conll-2000-1","name":"CoNLL-2000","full_name":"","num_papers_in_archive":8},{"url":"/dataset/hindencorp","name":"HindEnCorp","full_name":"","num_papers_in_archive":5}],"subtasks":[],"parent_tasks":[{"url":"/task/shallow-syntax","name":"Shallow Syntax"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":120,"tagged_in_all":447,"items":[{"url":"/paper/bidirectional-lstm-crf-models-for-sequence","title":"Bidirectional LSTM-CRF Models for Sequence Tagging","date":"2015-08-09","arxiv_id":"1508.01991","repositories_listed":25,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/optimal-hyperparameters-for-deep-lstm","title":"Optimal Hyperparameters for Deep LSTM-Networks for Sequence Labeling Tasks","date":"2017-07-21","arxiv_id":"1707.06799","repositories_listed":6,"syntology":null},{"url":"/paper/improving-named-entity-recognition-by","title":"Improving Named Entity Recognition by External Context Retrieving and Cooperative Learning","date":"2021-05-08","arxiv_id":"2105.03654","repositories_listed":3,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/ncrf-an-open-source-neural-sequence-labeling","title":"NCRF++: An Open-source Neural Sequence Labeling Toolkit","date":"2018-06-14","arxiv_id":"1806.05626","repositories_listed":3,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/semi-supervised-multitask-learning-for","title":"Semi-supervised Multitask Learning for Sequence Labeling","date":"2017-04-24","arxiv_id":"1704.07156","repositories_listed":3,"syntology":null},{"url":"/paper/bidirectional-decoding-improving-action","title":"Bidirectional Decoding: Improving Action Chunking via Guided Test-Time Sampling","date":"2024-08-30","arxiv_id":"2408.17355","repositories_listed":2,"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":3}},{"url":"/paper/def2vec-extensible-word-embeddings-from","title":"Def2Vec: Extensible Word Embeddings from Dictionary Definitions","date":"2023-12-16","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/building-odia-shallow-parser","title":"Building Odia Shallow Parser","date":"2022-04-19","arxiv_id":"2204.08960","repositories_listed":2,"syntology":null},{"url":"/paper/bertraffic-a-robust-bert-based-approach-for","title":"BERTraffic: BERT-based Joint Speaker Role and Speaker Change Detection for Air Traffic Control Communications","date":"2021-10-12","arxiv_id":"2110.05781","repositories_listed":2,"syntology":null},{"url":"/paper/automated-concatenation-of-embeddings-for-1","title":"Automated Concatenation of Embeddings for Structured Prediction","date":"2020-10-10","arxiv_id":"2010.05006","repositories_listed":2,"syntology":null},{"url":"/paper/joint-keyphrase-chunking-and-salience-ranking","title":"Capturing Global Informativeness in Open Domain Keyphrase Extraction","date":"2020-04-28","arxiv_id":"2004.13639","repositories_listed":2,"syntology":null},{"url":"/paper/design-challenges-and-misconceptions-in","title":"Design Challenges and Misconceptions in Neural Sequence Labeling","date":"2018-06-12","arxiv_id":"1806.04470","repositories_listed":2,"syntology":null},{"url":"/paper/a-joint-many-task-model-growing-a-neural","title":"A Joint Many-Task Model: Growing a Neural Network for Multiple NLP Tasks","date":"2016-11-05","arxiv_id":"1611.01587","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/natural-language-processing-almost-from","title":"Natural Language Processing (almost) from Scratch","date":"2011-03-02","arxiv_id":"1103.0398","repositories_listed":2,"syntology":null},{"url":"/paper/cast-enhancing-code-retrieval-augmented","title":"cAST: Enhancing Code Retrieval-Augmented Generation with Structural Chunking via Abstract Syntax Tree","date":"2025-06-18","arxiv_id":"2506.15655","repositories_listed":1,"syntology":null},{"url":"/paper/tablerag-a-retrieval-augmented-generation","title":"TableRAG: A Retrieval Augmented Generation Framework for Heterogeneous Document Reasoning","date":"2025-06-12","arxiv_id":"2506.10380","repositories_listed":1,"syntology":null},{"url":"/paper/real-time-execution-of-action-chunking-flow","title":"Real-Time Execution of Action Chunking Flow Policies","date":"2025-06-09","arxiv_id":"2506.07339","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/dynamic-chunking-and-selection-for-reading","title":"Dynamic Chunking and Selection for Reading Comprehension of Ultra-Long Context in Large Language Models","date":"2025-06-01","arxiv_id":"2506.00773","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_unverified":0,"n_pointer_only":9}},{"url":"/paper/context-is-gold-to-find-the-gold-passage","title":"Context is Gold to find the Gold Passage: Evaluating and Training Contextual Document Embeddings","date":"2025-05-30","arxiv_id":"2505.24782","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-chunk-size-for-long-document","title":"Rethinking Chunk Size For Long-Document Retrieval: A Multi-Dataset Analysis","date":"2025-05-27","arxiv_id":"2505.21700","repositories_listed":1,"syntology":null},{"url":"/paper/neusym-rag-hybrid-neural-symbolic-retrieval","title":"NeuSym-RAG: Hybrid Neural Symbolic Retrieval with Multiview Structuring for PDF Question Answering","date":"2025-05-26","arxiv_id":"2505.19754","repositories_listed":1,"syntology":null},{"url":"/paper/alto-adaptive-length-tokenizer-for","title":"ALTo: Adaptive-Length Tokenizer for Autoregressive Mask Generation","date":"2025-05-22","arxiv_id":"2505.16495","repositories_listed":1,"syntology":null},{"url":"/paper/reconstructing-context-evaluating-advanced","title":"Reconstructing Context: Evaluating Advanced Chunking Strategies for Retrieval-Augmented Generation","date":"2025-04-28","arxiv_id":"2504.19754","repositories_listed":1,"syntology":null},{"url":"/paper/flexchunk-enabling-100mx100m-out-of-core-spmv","title":"FlexChunk: Enabling 100M×100M Out-of-Core SpMV (~1.8 min, ~1.7 GB RAM) with Near-Linear Scaling","date":"2025-04-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/text-chunking-for-document-classification-for","title":"Text Chunking for Document Classification for Urban System Management using Large Language Models","date":"2025-03-31","arxiv_id":"2504.00274","repositories_listed":1,"syntology":null},{"url":"/paper/aistorian-lets-ai-be-a-historian-a-kg-powered","title":"AIstorian lets AI be a historian: A KG-powered multi-agent system for accurate biography generation","date":"2025-03-14","arxiv_id":"2503.11346","repositories_listed":1,"syntology":null},{"url":"/paper/moc-mixtures-of-text-chunking-learners-for","title":"MoC: Mixtures of Text Chunking Learners for Retrieval-Augmented Generation System","date":"2025-03-12","arxiv_id":"2503.09600","repositories_listed":1,"syntology":null},{"url":"/paper/timeloc-a-unified-end-to-end-framework-for","title":"TimeLoc: A Unified End-to-End Framework for Precise Timestamp Localization in Long Videos","date":"2025-03-09","arxiv_id":"2503.06526","repositories_listed":1,"syntology":null},{"url":"/paper/kidneytalk-open-no-code-deployment-of-a","title":"KidneyTalk-open: No-code Deployment of a Private Large Language Model with Medical Documentation-Enhanced Knowledge Database for Kidney Disease","date":"2025-03-06","arxiv_id":"2503.04153","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-vision-language-action-models","title":"Fine-Tuning Vision-Language-Action Models: Optimizing Speed and Success","date":"2025-02-27","arxiv_id":"2502.19645","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}