{"url":"/task/text-annotation","name":"text annotation","slug":"text-annotation","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":127,"papers_with_code":39,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/oasst1","name":"OASST1","full_name":"OpenAssistant Conversations Dataset","num_papers_in_archive":25},{"url":"/dataset/mathwell-human-annotation-dataset","name":"MATHWELL Human Annotation Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/mibot-motivational-interviewing-for-smoking","name":"MIBot - Motivational Interviewing for Smoking Cessation Dataset, based on MIBOT Version 6.3A","full_name":"","num_papers_in_archive":1},{"url":"/dataset/reddit-posts-related-to-eating-disorders-and","name":"Reddit Posts Related To Eating Disorders and Dieting","full_name":"Topic Annotations on Reddit Posts from Eating Disorders and Dieting Forums by Human and LLMs","num_papers_in_archive":1},{"url":"/dataset/vmd","name":"VMD","full_name":"Virtual Moderation Dataset","num_papers_in_archive":1},{"url":"/dataset/comentarios-vacuna-vph","name":"Text_VPH","full_name":"","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":39,"tagged_in_all":127,"items":[{"url":"/paper/a-bilingual-openworld-video-text-dataset-and","title":"A Bilingual, OpenWorld Video Text Dataset and End-to-end Video Text Spotter with Transformer","date":"2021-12-09","arxiv_id":"2112.04888","repositories_listed":3,"syntology":{"n":19,"n_ran":16,"n_unverified":3,"n_pointer_only":19}},{"url":"/paper/text-guided-foundation-model-adaptation-for","title":"Text-guided Foundation Model Adaptation for Pathological Image Classification","date":"2023-07-27","arxiv_id":"2307.14901","repositories_listed":2,"syntology":null},{"url":"/paper/how-to-use-llms-for-text-analysis","title":"How to use LLMs for Text Analysis","date":"2023-07-24","arxiv_id":"2307.13106","repositories_listed":2,"syntology":null},{"url":"/paper/patent-sentiment-analysis-to-highlight-patent","title":"Patent Sentiment Analysis to Highlight Patent Paragraphs","date":"2021-11-06","arxiv_id":"2111.09741","repositories_listed":2,"syntology":null},{"url":"/paper/open-images-v5-text-annotation-and-yet","title":"Open Images V5 Text Annotation and Yet Another Mask Text Spotter","date":"2021-06-23","arxiv_id":"2106.12326","repositories_listed":2,"syntology":null},{"url":"/paper/marvin-semantic-annotation-using-multiple","title":"Marvin: Semantic annotation using multiple knowledge sources","date":"2016-02-01","arxiv_id":"1602.00515","repositories_listed":2,"syntology":null},{"url":"/paper/educoder-an-open-source-annotation-system-for","title":"EduCoder: An Open-Source Annotation System for Education Transcript Data","date":"2025-07-07","arxiv_id":"2507.05385","repositories_listed":1,"syntology":null},{"url":"/paper/probably-approximately-correct-labels","title":"Probably Approximately Correct Labels","date":"2025-06-12","arxiv_id":"2506.10908","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":4}},{"url":"/paper/avatarshield-visual-reinforcement-learning","title":"AvatarShield: Visual Reinforcement Learning for Human-Centric Video Forgery Detection","date":"2025-05-21","arxiv_id":"2505.15173","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-llm-text-annotation-skills-a-study","title":"Comparing LLM Text Annotation Skills: A Study on Human Rights Violations in Social Media Data","date":"2025-05-15","arxiv_id":"2505.10260","repositories_listed":1,"syntology":null},{"url":"/paper/it-s-a-blind-match-towards-vision-language","title":"It's a (Blind) Match! Towards Vision-Language Correspondence without Parallel Data","date":"2025-03-31","arxiv_id":"2503.24129","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/gpt4scene-understand-3d-scenes-from-videos","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","date":"2025-01-02","arxiv_id":"2501.01428","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/holmes-vau-towards-long-term-video-anomaly","title":"Holmes-VAU: Towards Long-term Video Anomaly Understanding at Any Granularity","date":"2024-12-09","arxiv_id":"2412.06171","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/verified-a-video-corpus-moment-retrieval","title":"VERIFIED: A Video Corpus Moment Retrieval Benchmark for Fine-Grained Video Understanding","date":"2024-10-11","arxiv_id":"2410.08593","repositories_listed":1,"syntology":null},{"url":"/paper/2408-03125","title":"COMMENTATOR: A Code-mixed Multilingual Text Annotation Framework","date":"2024-08-06","arxiv_id":"2408.03125","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/busclean-open-source-software-for-breast","title":"BUSClean: Open-source software for breast ultrasound image pre-processing and knowledge extraction for medical AI","date":"2024-07-16","arxiv_id":"2407.11316","repositories_listed":1,"syntology":null},{"url":"/paper/chatbots-are-not-reliable-text-annotators","title":"Chatbots Are Not Reliable Text Annotators","date":"2023-11-09","arxiv_id":"2311.05769","repositories_listed":1,"syntology":null},{"url":"/paper/coannotating-uncertainty-guided-work","title":"CoAnnotating: Uncertainty-Guided Work Allocation between Human and Large Language Models for Data Annotation","date":"2023-10-24","arxiv_id":"2310.15638","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/llm-in-the-loop-leveraging-large-language","title":"LLM-in-the-loop: Leveraging Large Language Model for Thematic Analysis","date":"2023-10-23","arxiv_id":"2310.15100","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/motion-x-a-large-scale-3d-expressive-whole","title":"Motion-X: A Large-scale 3D Expressive Whole-body Human Motion Dataset","date":"2023-07-03","arxiv_id":"2307.00818","repositories_listed":1,"syntology":null},{"url":"/paper/high-fidelity-3d-face-generation-from-natural","title":"High-Fidelity 3D Face Generation from Natural Language Descriptions","date":"2023-05-05","arxiv_id":"2305.03302","repositories_listed":1,"syntology":null},{"url":"/paper/chatgpt-outperforms-crowd-workers-for-text","title":"ChatGPT Outperforms Crowd-Workers for Text-Annotation Tasks","date":"2023-03-27","arxiv_id":"2303.15056","repositories_listed":1,"syntology":null},{"url":"/paper/potato-the-portable-text-annotation-tool","title":"POTATO: The Portable Text Annotation Tool","date":"2022-12-16","arxiv_id":"2212.08620","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/measuring-annotator-agreement-generally","title":"Measuring Annotator Agreement Generally across Complex Structured, Multi-object, and Free-text Annotation Tasks","date":"2022-12-15","arxiv_id":"2212.09503","repositories_listed":1,"syntology":null},{"url":"/paper/freevc-towards-high-quality-text-free-one","title":"FreeVC: Towards High-Quality Text-Free One-Shot Voice Conversion","date":"2022-10-27","arxiv_id":"2210.15418","repositories_listed":1,"syntology":null},{"url":"/paper/humset-dataset-of-multilingual-information","title":"HumSet: Dataset of Multilingual Information Extraction and Classification for Humanitarian Crisis Response","date":"2022-10-10","arxiv_id":"2210.04573","repositories_listed":1,"syntology":null},{"url":"/paper/sciannotate-a-tool-for-integrating-weak","title":"SciAnnotate: A Tool for Integrating Weak Labeling Sources for Sequence Labeling","date":"2022-08-07","arxiv_id":"2208.10241","repositories_listed":1,"syntology":null},{"url":"/paper/lvit-language-meets-vision-transformer-in","title":"LViT: Language meets Vision Transformer in Medical Image Segmentation","date":"2022-06-29","arxiv_id":"2206.14718","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/fine-grained-image-captioning-with-clip","title":"Fine-grained Image Captioning with CLIP Reward","date":"2022-05-26","arxiv_id":"2205.13115","repositories_listed":1,"syntology":null},{"url":"/paper/dotat-a-domain-oriented-text-annotation-tool-1","title":"DoTAT: A Domain-oriented Text Annotation Tool","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}