{"url":"/task/semantic-entity-labeling","name":"Semantic entity labeling","slug":"semantic-entity-labeling","description_markdown":"- One of Form Understanding task (Word grouping, Semantic entity labeling, Entity linking)\r\n- Classifying entities into one of four pre-defined categories: question, answer, header and, other.\r\n\r\ncited from\r\n\r\nG. Jaume, H. K. Ekenel, J. Thiran \"FUNSD: A Dataset for Form Understanding in Noisy Scanned Documents,\" 2019","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":14,"papers_with_code":12,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":2,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/semantic-entity-labeling-on-funsd","slug":"semantic-entity-labeling-on-funsd","dataset":"FUNSD","dataset_url":"/dataset/funsd","rows_in_archive":15,"metrics":["F1"],"first_row_in_archive_order":{"model":"LayoutMask (large)","paper_title":"LayoutMask: Enhance Text-Layout Interaction in Multi-modal Pre-training for Document Understanding","paper_url":"/paper/layoutmask-enhance-text-layout-interaction-in","paper_date":"2023-05-30","arxiv_id":"2305.18721","code_links":[],"syntology":null}},{"leaderboard":"/sota/semantic-entity-labeling-on-ec-funsd","slug":"semantic-entity-labeling-on-ec-funsd","dataset":"EC-FUNSD","dataset_url":"/dataset/ec-funsd","rows_in_archive":8,"metrics":["F1"],"first_row_in_archive_order":{"model":"RORE (LayoutLMv3-large)","paper_title":"Modeling Layout Reading Order as Ordering Relations for Visually-rich Document Understanding","paper_url":"/paper/modeling-layout-reading-order-as-ordering","paper_date":"2024-09-29","arxiv_id":"2409.19672","code_links":[{"title":"chongzhangFDU/ROOR","url":"https://github.com/chongzhangFDU/ROOR"}],"syntology":null}}],"datasets":[{"url":"/dataset/funsd","name":"FUNSD","full_name":"Form Understanding in Noisy Scanned Documents","num_papers_in_archive":179},{"url":"/dataset/ec-funsd","name":"EC-FUNSD","full_name":"","num_papers_in_archive":3}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":12,"of":12,"tagged_in_all":14,"items":[{"url":"/paper/layoutlmv2-multi-modal-pre-training-for","title":"LayoutLMv2: Multi-modal Pre-training for Visually-Rich Document Understanding","date":"2020-12-29","arxiv_id":"2012.14740","repositories_listed":9,"syntology":null},{"url":"/paper/lilt-a-simple-yet-effective-language","title":"LiLT: A Simple yet Effective Language-Independent Layout Transformer for Structured Document Understanding","date":"2022-02-28","arxiv_id":"2202.13669","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/reading-order-matters-information-extraction","title":"Reading Order Matters: Information Extraction from Visually-rich Documents by Token Path Prediction","date":"2023-10-17","arxiv_id":"2310.11016","repositories_listed":2,"syntology":null},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","arxiv_id":"2210.06155","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/modeling-layout-reading-order-as-ordering","title":"Modeling Layout Reading Order as Ordering Relations for Visually-rich Document Understanding","date":"2024-09-29","arxiv_id":"2409.19672","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-the-evaluation-of-pre-trained-text","title":"Rethinking the Evaluation of Pre-trained Text-and-Layout Models from an Entity-Centric Perspective","date":"2024-02-04","arxiv_id":"2402.02379","repositories_listed":1,"syntology":null},{"url":"/paper/peneo-unifying-line-extraction-line-grouping","title":"PEneo: Unifying Line Extraction, Line Grouping, and Entity Linking for End-to-end Document Pair Extraction","date":"2024-01-07","arxiv_id":"2401.03472","repositories_listed":1,"syntology":null},{"url":"/paper/geolayoutlm-geometric-pre-training-for-visual","title":"GeoLayoutLM: Geometric Pre-training for Visual Information Extraction","date":"2023-04-21","arxiv_id":"2304.10759","repositories_listed":1,"syntology":null},{"url":"/paper/structextv2-masked-visual-textual-prediction","title":"StrucTexTv2: Masked Visual-Textual Prediction for Document Image Pre-training","date":"2023-03-01","arxiv_id":"2303.00289","repositories_listed":1,"syntology":null},{"url":"/paper/xdoc-unified-pre-training-for-cross-format","title":"XDoc: Unified Pre-training for Cross-Format Document Understanding","date":"2022-10-06","arxiv_id":"2210.02849","repositories_listed":1,"syntology":null},{"url":"/paper/doc2graph-a-task-agnostic-document","title":"Doc2Graph: a Task Agnostic Document Understanding Framework based on Graph Neural Networks","date":"2022-08-23","arxiv_id":"2208.11168","repositories_listed":1,"syntology":null}],"syntology_records":2,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}