{"url":"/task/object-discovery","name":"Object Discovery","slug":"object-discovery","description_markdown":"**Object Discovery** is the task of identifying previously unseen objects.\n\n\n<span class=\"description-source\">Source: [Unsupervised Object Discovery and Segmentation of RGBD-images ](https://arxiv.org/abs/1710.06929)</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":210,"papers_with_code":96,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/uvo","name":"UVO","full_name":"Unidentified Video Objects: A Benchmark for Dense, Open-World Segmentation","num_papers_in_archive":27},{"url":"/dataset/shapestacks","name":"ShapeStacks","full_name":"","num_papers_in_archive":22},{"url":"/dataset/infinity-mm","name":"Infinity-MM","full_name":"","num_papers_in_archive":5}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":96,"tagged_in_all":210,"items":[{"url":"/paper/object-centric-learning-with-slot-attention","title":"Object-Centric Learning with Slot Attention","date":"2020-06-26","arxiv_id":"2006.15055","repositories_listed":8,"syntology":{"n":21,"n_ran":14,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/vision-transformers-need-registers","title":"Vision Transformers Need Registers","date":"2023-09-28","arxiv_id":"2309.16588","repositories_listed":6,"syntology":{"n":20,"n_ran":4,"n_unverified":16,"n_pointer_only":2}},{"url":"/paper/learning-open-world-object-proposals-without","title":"Learning Open-World Object Proposals without Learning to Classify","date":"2021-08-15","arxiv_id":"2108.06753","repositories_listed":6,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/monet-unsupervised-scene-decomposition-and","title":"MONet: Unsupervised Scene Decomposition and Representation","date":"2019-01-22","arxiv_id":"1901.11390","repositories_listed":5,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/learn-to-pay-attention","title":"Learn To Pay Attention","date":"2018-04-06","arxiv_id":"1804.02391","repositories_listed":4,"syntology":null},{"url":"/paper/guesswhat-visual-object-discovery-through","title":"GuessWhat?! Visual object discovery through multi-modal dialogue","date":"2016-11-23","arxiv_id":"1611.08481","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/towards-robust-robot-3d-perception-in-urban","title":"Towards Robust Robot 3D Perception in Urban Environments: The UT Campus Object Dataset","date":"2023-09-24","arxiv_id":"2309.13549","repositories_listed":3,"syntology":null},{"url":"/paper/cuvler-enhanced-unsupervised-object","title":"CuVLER: Enhanced Unsupervised Object Discoveries through Exhaustive Self-Supervised Transformers","date":"2024-03-12","arxiv_id":"2403.07700","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/object-discovery-from-motion-guided-tokens","title":"Object Discovery from Motion-Guided Tokens","date":"2023-03-27","arxiv_id":"2303.15555","repositories_listed":2,"syntology":{"n":29,"n_ran":18,"n_unverified":11,"n_pointer_only":29}},{"url":"/paper/invariant-slot-attention-object-discovery","title":"Invariant Slot Attention: Object Discovery with Slot-Centric Reference Frames","date":"2023-02-09","arxiv_id":"2302.04973","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-domain-adaptation-for-nighttime","title":"Unsupervised Domain Adaptation for Nighttime Aerial Tracking","date":"2022-03-20","arxiv_id":"2203.10541","repositories_listed":2,"syntology":{"n":11,"n_ran":4,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/what-matters-for-meta-learning-vision","title":"What Matters For Meta-Learning Vision Regression Tasks?","date":"2022-03-09","arxiv_id":"2203.04905","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-image-decomposition-with-phase","title":"Unsupervised Image Decomposition with Phase-Correlation Networks","date":"2021-10-07","arxiv_id":"2110.03473","repositories_listed":2,"syntology":null},{"url":"/paper/localizing-objects-with-self-supervised","title":"Localizing Objects with Self-Supervised Transformers and no Labels","date":"2021-09-29","arxiv_id":"2109.14279","repositories_listed":2,"syntology":null},{"url":"/paper/genesis-generative-scene-inference-and","title":"GENESIS: Generative Scene Inference and Sampling with Object-Centric Latent Representations","date":"2019-07-30","arxiv_id":"1907.13052","repositories_listed":2,"syntology":null},{"url":"/paper/cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":0,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/efficient-dialog-policy-learning-via-positive","title":"Efficient Dialog Policy Learning via Positive Memory Retention","date":"2018-10-02","arxiv_id":"1810.01371","repositories_listed":2,"syntology":null},{"url":"/paper/when-does-pruning-benefit-vision","title":"When Does Pruning Benefit Vision Representations?","date":"2025-07-02","arxiv_id":"2507.01722","repositories_listed":1,"syntology":null},{"url":"/paper/binding-threshold-units-with-artificial","title":"Binding threshold units with artificial oscillatory neurons","date":"2025-05-06","arxiv_id":"2505.03648","repositories_listed":1,"syntology":null},{"url":"/paper/are-we-done-with-object-centric-learning","title":"Are We Done with Object-Centric Learning?","date":"2025-04-09","arxiv_id":"2504.07092","repositories_listed":1,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/vector-quantized-vision-foundation-models-for","title":"Vector-Quantized Vision Foundation Models for Object-Centric Learning","date":"2025-02-27","arxiv_id":"2502.20263","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/frontiernet-learning-visual-cues-to-explore","title":"FrontierNet: Learning Visual Cues to Explore","date":"2025-01-08","arxiv_id":"2501.04597","repositories_listed":1,"syntology":null},{"url":"/paper/pickscan-object-discovery-and-reconstruction","title":"PickScan: Object discovery and reconstruction from handheld interactions","date":"2024-11-17","arxiv_id":"2411.11196","repositories_listed":1,"syntology":null},{"url":"/paper/grouped-discrete-representation-for-object","title":"Grouped Discrete Representation for Object-Centric Learning","date":"2024-11-04","arxiv_id":"2411.02299","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/artificial-kuramoto-oscillatory-neurons","title":"Artificial Kuramoto Oscillatory Neurons","date":"2024-10-17","arxiv_id":"2410.13821","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/peekaboo-hiding-parts-of-an-image-for","title":"PEEKABOO: Hiding parts of an image for unsupervised object localization","date":"2024-07-24","arxiv_id":"2407.17628","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-transfer-with-simulated-inter-image","title":"Knowledge Transfer with Simulated Inter-Image Erasing for Weakly Supervised Semantic Segmentation","date":"2024-07-03","arxiv_id":"2407.02768","repositories_listed":1,"syntology":null},{"url":"/paper/dipex-dispersing-prompt-expansion-for-class","title":"DiPEx: Dispersing Prompt Expansion for Class-Agnostic Object Detection","date":"2024-06-21","arxiv_id":"2406.14924","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/adaptive-slot-attention-object-discovery-with-1","title":"Adaptive Slot Attention: Object Discovery with Dynamic Slot Number","date":"2024-06-13","arxiv_id":"2406.09196","repositories_listed":1,"syntology":null},{"url":"/paper/collaborative-novel-object-discovery-and-box","title":"Collaborative Novel Object Discovery and Box-Guided Cross-Modal Alignment for Open-Vocabulary 3D Object Detection","date":"2024-06-02","arxiv_id":"2406.00830","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}