{"url":"/dataset/tnl2k","name":"TNL2K","full_name":"Tracking by natural language","description_markdown":"**Tracking by Natural Language** (**TNL2K**) is constructed for the evaluation of tracking by natural language specification. TNL2K features:\r\n\r\n- Large-scale: 2,000 sequences, contains 1,244,340 frames, 663 words, 1300 / 700 for the train / testing respectively \r\n\r\n- High-quality: Manual annotation with careful inspection in each frame \r\n\r\n- Multi-modal: Providing visual and language annotation for each sequence \r\n\r\n- Adversarial-samples: Randomly adding adversarial samples for research on adversarial attack and defence \r\n\r\n- Significant-appearance-variation:  Containing videos with cloth/face change for pedestrian \r\n\r\n- Heterogeneous: Containing RGB, thermal, Cartoon,  Synthetic data \r\n\r\n- Multiple-baseline: Tracking-by-BBox, Tracking-by-Language, Tracking-by-Joint-BBox-Language\r\n\r\nSource: [Towards More Flexible and Accurate Object Tracking with Natural Language:  Algorithms and Benchmark](https://sites.google.com/view/langtrackbenchmark/)","description_withheld":null,"homepage":"https://sites.google.com/view/langtrackbenchmark/","introduced_date":"2021-03-31","introduced_date_note":null,"introduced_by":{"paper":"/paper/towards-more-flexible-and-accurate-object","title":"Towards More Flexible and Accurate Object Tracking with Natural Language: Algorithms and Benchmark","first_author":"Xiao Wang","url":null},"license":{"name":"Custom","url":"https://sites.google.com/view/langtrackbenchmark/"},"modalities":[],"tasks":[{"name":"Visual Object Tracking","url":"/task/visual-object-tracking","datasets_with_task":"/datasets/task/visual-object-tracking"},{"name":"Visual Tracking","url":"/task/visual-tracking","datasets_with_task":"/datasets/task/visual-tracking"}],"languages":[],"variants":["TNL2K"],"data_loaders":[],"num_papers_in_archive":62,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/visual-object-tracking-on-tnl2k","task":"Visual Object Tracking","dataset_variant":"TNL2K","rows":16,"metrics":["AUC","precision","Normalized Precision"],"first_row_in_archive_order":{"model":"MCITrack-L384","paper":"/paper/exploring-enhanced-contextual-information-for-1","metrics":{"AUC":"65.3"},"code_links":[{"title":"kangben258/MCITrack","url":"https://github.com/kangben258/MCITrack"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/visual-tracking-on-tnl2k","task":"Visual Tracking","dataset_variant":"TNL2K","rows":6,"metrics":["AUC","precision"],"first_row_in_archive_order":{"model":"ARTrack-L","paper":"/paper/autoregressive-visual-tracking","metrics":{"AUC":"60.3"},"code_links":[{"title":"miv-xjtu/artrack","url":"https://github.com/miv-xjtu/artrack"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/spmtrack-spatio-temporal-parameter-efficient","title":"SPMTrack: Spatio-Temporal Parameter-Efficient Fine-Tuning with Mixture of Experts for Scalable Visual Tracking","date":"2025-03-24","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/exploring-enhanced-contextual-information-for-1","title":"Exploring Enhanced Contextual Information for Video-Level Object Tracking","date":"2024-12-15","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/rtracker-recoverable-tracking-via-pn-tree","title":"RTracker: Recoverable Tracking via PN Tree Structured Memory","date":"2024-03-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tracking-meets-lora-faster-training-larger","title":"Tracking Meets LoRA: Faster Training, Larger Model, Stronger Performance","date":"2024-03-08","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":5,"samples_ran":3,"samples_unverified":2,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/odtrack-online-dense-temporal-token-learning","title":"ODTrack: Online Dense Temporal Token Learning for Visual Tracking","date":"2024-01-03","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":1,"samples_unverified":0,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/artrackv2-prompting-autoregressive-tracker","title":"ARTrackV2: Prompting Autoregressive Tracker Where to Look and How to Describe","date":"2023-12-28","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/seqtrack-sequence-to-sequence-learning-for","title":"Unified Sequence-to-Sequence Learning for Single- and Multi-Modal Visual Object Tracking","date":"2023-04-27","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/dropmae-masked-autoencoders-with-spatial","title":"DropMAE: Masked Autoencoders with Spatial-Attention Dropout for Tracking Tasks","date":"2023-04-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/joint-visual-grounding-and-tracking-with","title":"Joint Visual Grounding and Tracking with Natural Language Specification","date":"2023-03-21","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":1,"samples_unverified":3,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":4,"samples_ran":3,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/autoregressive-visual-tracking","title":"Autoregressive Visual Tracking","date":"2023-01-01","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/joint-feature-learning-and-relation-modeling","title":"Joint Feature Learning and Relation Modeling for Tracking: A One-Stream Framework","date":"2022-03-22","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":1,"samples_ran":0,"samples_unverified":1,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/towards-more-flexible-and-accurate-object","title":"Towards More Flexible and Accurate Object Tracking with Natural Language: Algorithms and Benchmark","date":"2021-03-31","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/2103-15436","title":"Transformer Tracking","date":"2021-03-29","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":3,"samples_ran":1,"samples_unverified":2,"pointer_only_for_licence":3,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":7,"samples_harvested":19,"samples_ran":9,"samples_unverified":10,"pointer_only_for_licence":3,"papers_with_no_sample_that_ran":2,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}