{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/4","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":18,"rows_per_page":100,"rows":[301,400],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/3","next":"/task/scene-understanding/papers/5","papers":[{"url":"/paper/doctr-disentangled-object-centric-transformer","slug":"doctr-disentangled-object-centric-transformer","title":"DOCTR: Disentangled Object-Centric Transformer for Point Scene Understanding","date":"2024-03-25","arxiv_id":"2403.16431","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-lidar-placements-for-robust","slug":"optimizing-lidar-placements-for-robust","title":"Is Your LiDAR Placement Optimized for 3D Scene Understanding?","date":"2024-03-25","arxiv_id":"2403.17009","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimizing-lidar-placements-for-robust#ran","syntology_url":"https://syntology.ai/paper/2403.17009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17009"}},"official":{"repos":["ywyeli/place3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autoinst-automatic-instance-based","slug":"autoinst-automatic-instance-based","title":"AutoInst: Automatic Instance-Based Segmentation of LiDAR 3D Scans","date":"2024-03-24","arxiv_id":"2403.16318","repositories_listed":1,"syntology":null},{"url":"/paper/volumetric-environment-representation-for","slug":"volumetric-environment-representation-for","title":"Volumetric Environment Representation for Vision-Language Navigation","date":"2024-03-21","arxiv_id":"2403.14158","repositories_listed":1,"syntology":null},{"url":"/paper/what-if-counterfactual-inception-to-mitigate","slug":"what-if-counterfactual-inception-to-mitigate","title":"What if...?: Thinking Counterfactual Keywords Helps to Mitigate Hallucination in Large Multi-modal Models","date":"2024-03-20","arxiv_id":"2403.13513","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-source-scale-bias-via-image","slug":"addressing-source-scale-bias-via-image","title":"Instance-Warp: Saliency Guided Image Warping for Unsupervised Domain Adaptation","date":"2024-03-19","arxiv_id":"2403.12712","repositories_listed":1,"syntology":null},{"url":"/paper/openocc-open-vocabulary-3d-scene","slug":"openocc-open-vocabulary-3d-scene","title":"OpenOcc: Open Vocabulary 3D Scene Reconstruction via Occupancy Representation","date":"2024-03-18","arxiv_id":"2403.11796","repositories_listed":1,"syntology":null},{"url":"/paper/omni-recon-towards-general-purpose-neural","slug":"omni-recon-towards-general-purpose-neural","title":"Omni-Recon: Harnessing Image-based Rendering for General-Purpose Neural Radiance Fields","date":"2024-03-17","arxiv_id":"2403.11131","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/omni-recon-towards-general-purpose-neural#ran","syntology_url":"https://syntology.ai/paper/2403.11131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11131"}},"official":{"repos":["GATECH-EIC/Omni-Recon"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/groupcontrast-semantic-aware-self-supervised","slug":"groupcontrast-semantic-aware-self-supervised","title":"GroupContrast: Semantic-aware Self-supervised Representation Learning for 3D Understanding","date":"2024-03-14","arxiv_id":"2403.09639","repositories_listed":1,"syntology":null},{"url":"/paper/moai-mixture-of-all-intelligence-for-large","slug":"moai-mixture-of-all-intelligence-for-large","title":"MoAI: Mixture of All Intelligence for Large Language and Vision Models","date":"2024-03-12","arxiv_id":"2403.07508","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moai-mixture-of-all-intelligence-for-large#ran","syntology_url":"https://syntology.ai/paper/2403.07508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07508"}},"official":{"repos":["ByungKwanLee/MoAI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-latent-graph-representations-of","slug":"optimizing-latent-graph-representations-of","title":"Optimizing Latent Graph Representations of Surgical Scenes for Zero-Shot Domain Transfer","date":"2024-03-11","arxiv_id":"2403.06953","repositories_listed":1,"syntology":null},{"url":"/paper/stealing-stable-diffusion-prior-for-robust","slug":"stealing-stable-diffusion-prior-for-robust","title":"Stealing Stable Diffusion Prior for Robust Monocular Depth Estimation","date":"2024-03-08","arxiv_id":"2403.05056","repositories_listed":1,"syntology":null},{"url":"/paper/embodied-understanding-of-driving-scenarios","slug":"embodied-understanding-of-driving-scenarios","title":"Embodied Understanding of Driving Scenarios","date":"2024-03-07","arxiv_id":"2403.04593","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/embodied-understanding-of-driving-scenarios#ran","syntology_url":"https://syntology.ai/paper/2403.04593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04593"}},"official":{"repos":["opendrivelab/elm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fusionvision-a-comprehensive-approach-of-3d","slug":"fusionvision-a-comprehensive-approach-of-3d","title":"FusionVision: A comprehensive approach of 3D object reconstruction and segmentation from RGB-D cameras using YOLO and fast segment anything","date":"2024-02-29","arxiv_id":"2403.00175","repositories_listed":1,"syntology":null},{"url":"/paper/one-model-to-use-them-all-training-a","slug":"one-model-to-use-them-all-training-a","title":"One model to use them all: Training a segmentation model with complementary datasets","date":"2024-02-29","arxiv_id":"2402.19340","repositories_listed":1,"syntology":null},{"url":"/paper/venvision3d-a-synthetic-perception-dataset","slug":"venvision3d-a-synthetic-perception-dataset","title":"WHU-Synthetic: A Synthetic Perception Dataset for 3-D Multitask Model Research","date":"2024-02-29","arxiv_id":"2402.19059","repositories_listed":1,"syntology":null},{"url":"/paper/avs-net-point-sampling-with-adaptive-voxel","slug":"avs-net-point-sampling-with-adaptive-voxel","title":"AVS-Net: Point Sampling with Adaptive Voxel Size for 3D Scene Understanding","date":"2024-02-27","arxiv_id":"2402.17521","repositories_listed":1,"syntology":null},{"url":"/paper/swin3d-effective-multi-source-pretraining-for","slug":"swin3d-effective-multi-source-pretraining-for","title":"Swin3D++: Effective Multi-Source Pretraining for 3D Indoor Scene Understanding","date":"2024-02-22","arxiv_id":"2402.14215","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swin3d-effective-multi-source-pretraining-for#ran","syntology_url":"https://syntology.ai/paper/2402.14215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14215"}},"official":{"repos":["microsoft/swin3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/semantically-aware-neural-radiance-fields-for","slug":"semantically-aware-neural-radiance-fields-for","title":"Semantically-aware Neural Radiance Fields for Visual Scene Understanding: A Comprehensive Review","date":"2024-02-17","arxiv_id":"2402.11141","repositories_listed":1,"syntology":null},{"url":"/paper/delving-into-multi-modal-multi-task","slug":"delving-into-multi-modal-multi-task","title":"Delving into Multi-modal Multi-task Foundation Models for Road Scene Understanding: From Learning Paradigm Perspectives","date":"2024-02-05","arxiv_id":"2402.02968","repositories_listed":1,"syntology":null},{"url":"/paper/sgs-slam-semantic-gaussian-splatting-for","slug":"sgs-slam-semantic-gaussian-splatting-for","title":"SGS-SLAM: Semantic Gaussian Splatting For Neural Dense SLAM","date":"2024-02-05","arxiv_id":"2402.03246","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sgs-slam-semantic-gaussian-splatting-for#ran","syntology_url":"https://syntology.ai/paper/2402.03246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03246"}},"official":{"repos":["shuhongll/sgs-slam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/non-central-panorama-indoor-dataset","slug":"non-central-panorama-indoor-dataset","title":"Non-central panorama indoor dataset","date":"2024-01-30","arxiv_id":"2401.17075","repositories_listed":1,"syntology":null},{"url":"/paper/towards-precise-3d-human-pose-estimation-with","slug":"towards-precise-3d-human-pose-estimation-with","title":"Towards Precise 3D Human Pose Estimation with Multi-Perspective Spatial-Temporal Relational Transformers","date":"2024-01-30","arxiv_id":"2401.16700","repositories_listed":1,"syntology":null},{"url":"/paper/unim-ov3d-uni-modality-open-vocabulary-3d","slug":"unim-ov3d-uni-modality-open-vocabulary-3d","title":"UniM-OV3D: Uni-Modality Open-Vocabulary 3D Scene Understanding with Fine-Grained Feature Representation","date":"2024-01-21","arxiv_id":"2401.11395","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/unim-ov3d-uni-modality-open-vocabulary-3d#ran","syntology_url":"https://syntology.ai/paper/2401.11395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11395"}},"official":{"repos":["hithqd/unim-ov3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/icgnet-a-unified-approach-for-instance","slug":"icgnet-a-unified-approach-for-instance","title":"ICGNet: A Unified Approach for Instance-Centric Grasping","date":"2024-01-18","arxiv_id":"2401.09939","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/icgnet-a-unified-approach-for-instance#ran","syntology_url":"https://syntology.ai/paper/2401.09939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09939"}},"official":{"repos":["renezurbruegg/icg_benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/garfield-group-anything-with-radiance-fields","slug":"garfield-group-anything-with-radiance-fields","title":"GARField: Group Anything with Radiance Fields","date":"2024-01-17","arxiv_id":"2401.09419","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/garfield-group-anything-with-radiance-fields#ran","syntology_url":"https://syntology.ai/paper/2401.09419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09419"}},"official":{"repos":["chungmin99/garfield"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rsud20k-a-dataset-for-road-scene","slug":"rsud20k-a-dataset-for-road-scene","title":"RSUD20K: A Dataset for Road Scene Understanding In Autonomous Driving","date":"2024-01-14","arxiv_id":"2401.07322","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rsud20k-a-dataset-for-road-scene#ran","syntology_url":"https://syntology.ai/paper/2401.07322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07322"}},"official":{"repos":["hasibzunair/rsud20k"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/funnynet-w-multimodal-learning-of-funny","slug":"funnynet-w-multimodal-learning-of-funny","title":"FunnyNet-W: Multimodal Learning of Funny Moments in Videos in the Wild","date":"2024-01-08","arxiv_id":"2401.04210","repositories_listed":1,"syntology":null},{"url":"/paper/3dmit-3d-multi-modal-instruction-tuning-for","slug":"3dmit-3d-multi-modal-instruction-tuning-for","title":"3DMIT: 3D Multi-modal Instruction Tuning for Scene Understanding","date":"2024-01-06","arxiv_id":"2401.03201","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/3dmit-3d-multi-modal-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2401.03201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03201"}},"official":{"repos":["staymylove/3DMIT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/maplm-a-real-world-large-scale-vision","slug":"maplm-a-real-world-large-scale-vision","title":"MAPLM: A Real-World Large-Scale Vision-Language Benchmark for Map and Traffic Scene Understanding","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/embodiedscan-a-holistic-multi-modal-3d","slug":"embodiedscan-a-holistic-multi-modal-3d","title":"EmbodiedScan: A Holistic Multi-Modal 3D Perception Suite Towards Embodied AI","date":"2023-12-26","arxiv_id":"2312.16170","repositories_listed":1,"syntology":null},{"url":"/paper/di-v2x-learning-domain-invariant","slug":"di-v2x-learning-domain-invariant","title":"DI-V2X: Learning Domain-Invariant Representation for Vehicle-Infrastructure Collaborative 3D Object Detection","date":"2023-12-25","arxiv_id":"2312.15742","repositories_listed":1,"syntology":null},{"url":"/paper/wildscenes-a-benchmark-for-2d-and-3d-semantic","slug":"wildscenes-a-benchmark-for-2d-and-3d-semantic","title":"WildScenes: A Benchmark for 2D and 3D Semantic Segmentation in Large-scale Natural Environments","date":"2023-12-23","arxiv_id":"2312.15364","repositories_listed":1,"syntology":null},{"url":"/paper/pola4all-survey-of-polarimetric-applications","slug":"pola4all-survey-of-polarimetric-applications","title":"Pola4All: survey of polarimetric applications and an open-source toolkit to analyze polarization","date":"2023-12-22","arxiv_id":"2312.14697","repositories_listed":1,"syntology":null},{"url":"/paper/object-attribute-matters-in-visual-question","slug":"object-attribute-matters-in-visual-question","title":"Object Attribute Matters in Visual Question Answering","date":"2023-12-20","arxiv_id":"2401.09442","repositories_listed":1,"syntology":null},{"url":"/paper/open3dis-open-vocabulary-3d-instance","slug":"open3dis-open-vocabulary-3d-instance","title":"Open3DIS: Open-Vocabulary 3D Instance Segmentation with 2D Mask Guidance","date":"2023-12-17","arxiv_id":"2312.10671","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open3dis-open-vocabulary-3d-instance#ran","syntology_url":"https://syntology.ai/paper/2312.10671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10671"}},"official":{"repos":["VinAIResearch/Open3DIS"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simple-image-level-classification-improves","slug":"simple-image-level-classification-improves","title":"Simple Image-level Classification Improves Open-vocabulary Object Detection","date":"2023-12-16","arxiv_id":"2312.10439","repositories_listed":1,"syntology":null},{"url":"/paper/transformers-in-unsupervised-structure-from","slug":"transformers-in-unsupervised-structure-from","title":"Transformers in Unsupervised Structure-from-Motion","date":"2023-12-16","arxiv_id":"2312.10529","repositories_listed":1,"syntology":null},{"url":"/paper/living-scenes-multi-object-relocalization-and","slug":"living-scenes-multi-object-relocalization-and","title":"Living Scenes: Multi-object Relocalization and Reconstruction in Changing 3D Environments","date":"2023-12-14","arxiv_id":"2312.09138","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/living-scenes-multi-object-relocalization-and#ran","syntology_url":"https://syntology.ai/paper/2312.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09138"}},"official":{"repos":["GradientSpaces/LivingScenes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zoom-in-on-the-plant-fine-grained-analysis-of","slug":"zoom-in-on-the-plant-fine-grained-analysis-of","title":"Zoom in on the Plant: Fine-grained Analysis of Leaf, Stem and Vein Instances","date":"2023-12-14","arxiv_id":"2312.08805","repositories_listed":1,"syntology":null},{"url":"/paper/x4d-sceneformer-enhanced-scene-understanding","slug":"x4d-sceneformer-enhanced-scene-understanding","title":"X4D-SceneFormer: Enhanced Scene Understanding on 4D Point Cloud Videos through Cross-modal Knowledge Transfer","date":"2023-12-12","arxiv_id":"2312.07378","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x4d-sceneformer-enhanced-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2312.07378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07378"}},"official":{"repos":["jinglinglingling/x4d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-ss3d-diffusion-model-for-semi-1","slug":"diffusion-ss3d-diffusion-model-for-semi-1","title":"Diffusion-SS3D: Diffusion Model for Semi-supervised 3D Object Detection","date":"2023-12-05","arxiv_id":"2312.02966","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffusion-ss3d-diffusion-model-for-semi-1#ran","syntology_url":"https://syntology.ai/paper/2312.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02966"}},"official":{"repos":["luluho1208/diffusion-ss3d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/igfnet-illumination-guided-fusion-network-for","slug":"igfnet-illumination-guided-fusion-network-for","title":"IGFNet: Illumination-Guided Fusion Network for Semantic Scene Understanding using RGB-Thermal Images","date":"2023-12-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-grouping-segment-and-edit-anything","slug":"gaussian-grouping-segment-and-edit-anything","title":"Gaussian Grouping: Segment and Edit Anything in 3D Scenes","date":"2023-12-01","arxiv_id":"2312.00732","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gaussian-grouping-segment-and-edit-anything#ran","syntology_url":"https://syntology.ai/paper/2312.00732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00732"}},"official":{"repos":["lkeab/gaussian-grouping"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-label-efficient-3d-scene-parsing","slug":"generalized-label-efficient-3d-scene-parsing","title":"Generalized Robot 3D Vision-Language Model with Fast Rendering and Pre-Training Vision-Language Alignment","date":"2023-12-01","arxiv_id":"2312.00663","repositories_listed":1,"syntology":null},{"url":"/paper/language-embedded-3d-gaussians-for-open","slug":"language-embedded-3d-gaussians-for-open","title":"Language Embedded 3D Gaussians for Open-Vocabulary Scene Understanding","date":"2023-11-30","arxiv_id":"2311.18482","repositories_listed":1,"syntology":null},{"url":"/paper/sampro3d-locating-sam-prompts-in-3d-for-zero","slug":"sampro3d-locating-sam-prompts-in-3d-for-zero","title":"SAMPro3D: Locating SAM Prompts in 3D for Zero-Shot Scene Segmentation","date":"2023-11-29","arxiv_id":"2311.17707","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/sampro3d-locating-sam-prompts-in-3d-for-zero#ran","syntology_url":"https://syntology.ai/paper/2311.17707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17707"}},"official":{"repos":["GAP-LAB-CUHK-SZ/SAMPro3D"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-planar-reconstruction-with-feature","slug":"multi-task-planar-reconstruction-with-feature","title":"Multi-task Planar Reconstruction with Feature Warping Guidance","date":"2023-11-25","arxiv_id":"2311.14981","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-scene-graph-generation-with","slug":"enhancing-scene-graph-generation-with","title":"Enhancing Scene Graph Generation with Hierarchical Relationships and Commonsense Knowledge","date":"2023-11-21","arxiv_id":"2311.12889","repositories_listed":1,"syntology":null},{"url":"/paper/spectralgpt-spectral-foundation-model","slug":"spectralgpt-spectral-foundation-model","title":"SpectralGPT: Spectral Remote Sensing Foundation Model","date":"2023-11-13","arxiv_id":"2311.07113","repositories_listed":1,"syntology":null},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-road-with-gpt-4v-ision-early","slug":"on-the-road-with-gpt-4v-ision-early","title":"On the Road with GPT-4V(ision): Early Explorations of Visual-Language Model on Autonomous Driving","date":"2023-11-09","arxiv_id":"2311.05332","repositories_listed":1,"syntology":null},{"url":"/paper/tsp-transformer-task-specific-prompts-boosted","slug":"tsp-transformer-task-specific-prompts-boosted","title":"TSP-Transformer: Task-Specific Prompts Boosted Transformer for Holistic Scene Understanding","date":"2023-11-06","arxiv_id":"2311.03427","repositories_listed":1,"syntology":null},{"url":"/paper/neusyre-neuro-symbolic-visual-understanding","slug":"neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-of-unsupervised-monocular","slug":"continual-learning-of-unsupervised-monocular","title":"Continual Learning of Unsupervised Monocular Depth from Videos","date":"2023-11-04","arxiv_id":"2311.02393","repositories_listed":1,"syntology":null},{"url":"/paper/tpsence-towards-artifact-free-realistic-rain","slug":"tpsence-towards-artifact-free-realistic-rain","title":"TPSeNCE: Towards Artifact-Free Realistic Rain Generation for Deraining and Object Detection in Rain","date":"2023-11-01","arxiv_id":"2311.00660","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tpsence-towards-artifact-free-realistic-rain#ran","syntology_url":"https://syntology.ai/paper/2311.00660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00660"}},"official":{"repos":["shenzheng2000/tpsence"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/p2at-pyramid-pooling-axial-transformer-for","slug":"p2at-pyramid-pooling-axial-transformer-for","title":"P2AT: Pyramid Pooling Axial Transformer for Real-time Semantic Segmentation","date":"2023-10-23","arxiv_id":"2310.15025","repositories_listed":1,"syntology":null},{"url":"/paper/dualmlp-a-two-stream-fusion-model-for-3d","slug":"dualmlp-a-two-stream-fusion-model-for-3d","title":"DualMLP: a two-stream fusion model for 3D point cloud classification","date":"2023-10-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/talk2bev-language-enhanced-bird-s-eye-view","slug":"talk2bev-language-enhanced-bird-s-eye-view","title":"Talk2BEV: Language-enhanced Bird's-eye View Maps for Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.02251","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/talk2bev-language-enhanced-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2310.02251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02251"}},"official":null}},{"url":"/paper/transradar-adaptive-directional-transformer","slug":"transradar-adaptive-directional-transformer","title":"TransRadar: Adaptive-Directional Transformer for Real-Time Multi-View Radar Semantic Segmentation","date":"2023-10-03","arxiv_id":"2310.02260","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-visual-scene-understanding","slug":"adaptive-visual-scene-understanding","title":"Adaptive Visual Scene Understanding: Incremental Scene Graph Generation","date":"2023-10-02","arxiv_id":"2310.01636","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-visual-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2310.01636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01636"}},"official":{"repos":["zhanglab-deepneurocoglab/csegg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-dataset-for-localization-mapping","slug":"multimodal-dataset-for-localization-mapping","title":"Multimodal Dataset for Localization, Mapping and Crop Monitoring in Citrus Tree Farms","date":"2023-09-27","arxiv_id":"2309.15332","repositories_listed":1,"syntology":null},{"url":"/paper/shape-anchor-guided-holistic-indoor-scene","slug":"shape-anchor-guided-holistic-indoor-scene","title":"Shape Anchor Guided Holistic Indoor Scene Understanding","date":"2023-09-20","arxiv_id":"2309.11133","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shape-anchor-guided-holistic-indoor-scene#ran","syntology_url":"https://syntology.ai/paper/2309.11133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11133"}},"official":{"repos":["Geo-Tell/AncRec"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mask4d-end-to-end-mask-based-4d-panoptic","slug":"mask4d-end-to-end-mask-based-4d-panoptic","title":"Mask4D: End-to-End Mask-Based 4D Panoptic Segmentation for LiDAR Sequences","date":"2023-09-18","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi3drefer-grounding-text-description-to","slug":"multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi3drefer-grounding-text-description-to#ran","syntology_url":"https://syntology.ai/paper/2309.05251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05251"}},"official":{"repos":["3dlg-hcvc/M3DRef-CLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vote2cap-detr-decoupling-localization-and","slug":"vote2cap-detr-decoupling-localization-and","title":"Vote2Cap-DETR++: Decoupling Localization and Describing for End-to-End 3D Dense Captioning","date":"2023-09-06","arxiv_id":"2309.02999","repositories_listed":1,"syntology":null},{"url":"/paper/openins3d-snap-and-lookup-for-3d-open","slug":"openins3d-snap-and-lookup-for-3d-open","title":"OpenIns3D: Snap and Lookup for 3D Open-vocabulary Instance Segmentation","date":"2023-09-01","arxiv_id":"2309.00616","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openins3d-snap-and-lookup-for-3d-open#ran","syntology_url":"https://syntology.ai/paper/2309.00616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00616"}},"official":{"repos":["Pointcept/OpenIns3D"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-stage-factorized-spatio-temporal","slug":"multi-stage-factorized-spatio-temporal","title":"Multi-stage Factorized Spatio-Temporal Representation for RGB-D Action and Gesture Recognition","date":"2023-08-23","arxiv_id":"2308.12006","repositories_listed":1,"syntology":null},{"url":"/paper/summit-source-free-adaptation-of-uni-modal","slug":"summit-source-free-adaptation-of-uni-modal","title":"SUMMIT: Source-Free Adaptation of Uni-Modal Models to Multi-Modal Targets","date":"2023-08-23","arxiv_id":"2308.11880","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/summit-source-free-adaptation-of-uni-modal#ran","syntology_url":"https://syntology.ai/paper/2308.11880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11880"}},"official":{"repos":["csimo005/summit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-dark-scenes-by-contrasting","slug":"understanding-dark-scenes-by-contrasting","title":"Understanding Dark Scenes by Contrasting Multi-Modal Observations","date":"2023-08-23","arxiv_id":"2308.12320","repositories_listed":1,"syntology":null},{"url":"/paper/scannet-a-high-fidelity-dataset-of-3d-indoor","slug":"scannet-a-high-fidelity-dataset-of-3d-indoor","title":"ScanNet++: A High-Fidelity Dataset of 3D Indoor Scenes","date":"2023-08-22","arxiv_id":"2308.11417","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scannet-a-high-fidelity-dataset-of-3d-indoor#ran","syntology_url":"https://syntology.ai/paper/2308.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11417"}},"official":null}},{"url":"/paper/vision-relation-transformer-for-unbiased","slug":"vision-relation-transformer-for-unbiased","title":"Vision Relation Transformer for Unbiased Scene Graph Generation","date":"2023-08-18","arxiv_id":"2308.09472","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vision-relation-transformer-for-unbiased#ran","syntology_url":"https://syntology.ai/paper/2308.09472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09472"}},"official":{"repos":["visinf/veto"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/focusflow-boosting-key-points-optical-flow","slug":"focusflow-boosting-key-points-optical-flow","title":"FocusFlow: Boosting Key-Points Optical Flow Estimation for Autonomous Driving","date":"2023-08-14","arxiv_id":"2308.07104","repositories_listed":1,"syntology":null},{"url":"/paper/semantics-guided-transformer-based-sensor","slug":"semantics-guided-transformer-based-sensor","title":"Cognitive TransFuser: Semantics-guided Transformer-based Sensor Fusion for Improved Waypoint Prediction","date":"2023-08-04","arxiv_id":"2308.02126","repositories_listed":1,"syntology":null},{"url":"/paper/gated-driver-attention-predictor","slug":"gated-driver-attention-predictor","title":"Gated Driver Attention Predictor","date":"2023-08-01","arxiv_id":"2308.02530","repositories_listed":1,"syntology":null},{"url":"/paper/taskexpert-dynamically-assembling-multi-task","slug":"taskexpert-dynamically-assembling-multi-task","title":"TaskExpert: Dynamically Assembling Multi-Task Representations with Memorial Mixture-of-Experts","date":"2023-07-28","arxiv_id":"2307.15324","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/taskexpert-dynamically-assembling-multi-task#ran","syntology_url":"https://syntology.ai/paper/2307.15324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15324"}},"official":{"repos":["prismformore/multi-task-transformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/human-centric-scene-understanding-for-3d-1","slug":"human-centric-scene-understanding-for-3d-1","title":"Human-centric Scene Understanding for 3D Large-scale Scenarios","date":"2023-07-26","arxiv_id":"2307.14392","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-scene-understanding-for-3d-1#ran","syntology_url":"https://syntology.ai/paper/2307.14392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14392"}},"official":{"repos":["4dvlab/hucenlife"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-distillation-for-continual","slug":"revisiting-distillation-for-continual","title":"Revisiting Distillation for Continual Learning on Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-22","arxiv_id":"2307.12045","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-distillation-for-continual#ran","syntology_url":"https://syntology.ai/paper/2307.12045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12045"}},"official":{"repos":["longbai1006/cs-vqla"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cpcm-contextual-point-cloud-modeling-for","slug":"cpcm-contextual-point-cloud-modeling-for","title":"CPCM: Contextual Point Cloud Modeling for Weakly-supervised Point Cloud Semantic Segmentation","date":"2023-07-19","arxiv_id":"2307.10316","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cpcm-contextual-point-cloud-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2307.10316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10316"}},"official":{"repos":["lizhaoliu-Lec/CPCM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-open-vocabulary-detection-and","slug":"a-survey-on-open-vocabulary-detection-and","title":"A Survey on Open-Vocabulary Detection and Segmentation: Past, Present, and Future","date":"2023-07-18","arxiv_id":"2307.09220","repositories_listed":1,"syntology":null},{"url":"/paper/open-scene-understanding-grounded-situation","slug":"open-scene-understanding-grounded-situation","title":"Open Scene Understanding: Grounded Situation Recognition Meets Segment Anything for Helping People with Visual Impairments","date":"2023-07-15","arxiv_id":"2307.07757","repositories_listed":1,"syntology":null},{"url":"/paper/deepipcv2-lidar-powered-robust-environmental","slug":"deepipcv2-lidar-powered-robust-environmental","title":"DeepIPCv2: LiDAR-powered Robust Environmental Perception and Navigational Control for Autonomous Vehicle","date":"2023-07-13","arxiv_id":"2307.06647","repositories_listed":1,"syntology":null},{"url":"/paper/the-imptc-dataset-an-infrastructural-multi","slug":"the-imptc-dataset-an-infrastructural-multi","title":"The IMPTC Dataset: An Infrastructural Multi-Person Trajectory and Context Dataset","date":"2023-07-12","arxiv_id":"2307.06165","repositories_listed":1,"syntology":null},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-accurate-instance-segmentation-in","slug":"towards-accurate-instance-segmentation-in","title":"Towards accurate instance segmentation in large-scale LiDAR point clouds","date":"2023-07-06","arxiv_id":"2307.02877","repositories_listed":1,"syntology":null},{"url":"/paper/avsegformer-audio-visual-segmentation-with","slug":"avsegformer-audio-visual-segmentation-with","title":"AVSegFormer: Audio-Visual Segmentation with Transformer","date":"2023-07-03","arxiv_id":"2307.01146","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-surgical-instruments","slug":"generalizing-surgical-instruments","title":"Generalizing Surgical Instruments Segmentation to Unseen Domains with One-to-Many Synthesis","date":"2023-06-28","arxiv_id":"2306.16285","repositories_listed":1,"syntology":null},{"url":"/paper/towards-open-vocabulary-learning-a-survey","slug":"towards-open-vocabulary-learning-a-survey","title":"Towards Open Vocabulary Learning: A Survey","date":"2023-06-28","arxiv_id":"2306.15880","repositories_listed":1,"syntology":null},{"url":"/paper/ssc-rs-elevate-lidar-semantic-scene","slug":"ssc-rs-elevate-lidar-semantic-scene","title":"SSC-RS: Elevate LiDAR Semantic Scene Completion with Representation Separation and BEV Fusion","date":"2023-06-27","arxiv_id":"2306.15349","repositories_listed":1,"syntology":null},{"url":"/paper/openmask3d-open-vocabulary-3d-instance","slug":"openmask3d-open-vocabulary-3d-instance","title":"OpenMask3D: Open-Vocabulary 3D Instance Segmentation","date":"2023-06-23","arxiv_id":"2306.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openmask3d-open-vocabulary-3d-instance#ran","syntology_url":"https://syntology.ai/paper/2306.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13631"}},"official":{"repos":["OpenMask3D/openmask3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-view-3d-object-reconstruction-and","slug":"multi-view-3d-object-reconstruction-and","title":"Multi-view 3D Object Reconstruction and Uncertainty Modelling with Neural Shape Prior","date":"2023-06-17","arxiv_id":"2306.11739","repositories_listed":1,"syntology":null},{"url":"/paper/panoocc-unified-occupancy-representation-for","slug":"panoocc-unified-occupancy-representation-for","title":"PanoOcc: Unified Occupancy Representation for Camera-based 3D Panoptic Segmentation","date":"2023-06-16","arxiv_id":"2306.10013","repositories_listed":1,"syntology":null},{"url":"/paper/estimating-generic-3d-room-structures-from-2d-1","slug":"estimating-generic-3d-room-structures-from-2d-1","title":"Estimating Generic 3D Room Structures from 2D Annotations","date":"2023-06-15","arxiv_id":"2306.09077","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/estimating-generic-3d-room-structures-from-2d-1#ran","syntology_url":"https://syntology.ai/paper/2306.09077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09077"}},"official":{"repos":["google-research/cad-estate"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/invpt-inverted-pyramid-multi-task-transformer","slug":"invpt-inverted-pyramid-multi-task-transformer","title":"InvPT++: Inverted Pyramid Multi-Task Transformer for Visual Scene Understanding","date":"2023-06-08","arxiv_id":"2306.04842","repositories_listed":1,"syntology":null},{"url":"/paper/snap-self-supervised-neural-maps-for-visual-1","slug":"snap-self-supervised-neural-maps-for-visual-1","title":"SNAP: Self-Supervised Neural Maps for Visual Positioning and Semantic Understanding","date":"2023-06-08","arxiv_id":"2306.05407","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/snap-self-supervised-neural-maps-for-visual-1#ran","syntology_url":"https://syntology.ai/paper/2306.05407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05407"}},"official":{"repos":["google-research/snap"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-label-free-scene-understanding-by","slug":"towards-label-free-scene-understanding-by","title":"Towards Label-free Scene Understanding by Vision Foundation Models","date":"2023-06-06","arxiv_id":"2306.03899","repositories_listed":1,"syntology":null},{"url":"/paper/towards-in-context-scene-understanding","slug":"towards-in-context-scene-understanding","title":"Towards In-context Scene Understanding","date":"2023-06-02","arxiv_id":"2306.01667","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-in-context-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2306.01667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01667"}},"official":null}},{"url":"/paper/point-gcc-universal-self-supervised-3d-scene","slug":"point-gcc-universal-self-supervised-3d-scene","title":"Point-GCC: Universal Self-supervised 3D Scene Pre-training via Geometry-Color Contrast","date":"2023-05-31","arxiv_id":"2305.19623","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-vision-transformers-for-3d","slug":"self-supervised-vision-transformers-for-3d","title":"Self-supervised Vision Transformers for 3D Pose Estimation of Novel Objects","date":"2023-05-31","arxiv_id":"2306.00129","repositories_listed":1,"syntology":null}],"record_sha256":"bfcaf0f0569e651ab6a832b4a7c57cfa7d445de8bc8be16a388c8fe76fff3261","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}