{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/2","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":18,"rows_per_page":100,"rows":[101,200],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding","next":"/task/scene-understanding/papers/3","papers":[{"url":"/paper/alfworld-aligning-text-and-embodied","slug":"alfworld-aligning-text-and-embodied","title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","date":"2020-10-08","arxiv_id":"2010.03768","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":4,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"11 ran (of which 4 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/alfworld-aligning-text-and-embodied#ran","syntology_url":"https://syntology.ai/paper/2010.03768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.03768"}},"official":{"repos":["alfworld/alfworld"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":4,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-semantic-segmentation-of-urban-scale","slug":"towards-semantic-segmentation-of-urban-scale","title":"Towards Semantic Segmentation of Urban-Scale 3D Point Clouds: A Dataset, Benchmarks and Challenges","date":"2020-09-07","arxiv_id":"2009.03137","repositories_listed":2,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/towards-semantic-segmentation-of-urban-scale#ran","syntology_url":"https://syntology.ai/paper/2009.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.03137"}},"official":{"repos":["QingyongHu/SensatUrban"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/polysemy-deciphering-network-for-robust-human","slug":"polysemy-deciphering-network-for-robust-human","title":"Polysemy Deciphering Network for Robust Human-Object Interaction Detection","date":"2020-08-07","arxiv_id":"2008.02918","repositories_listed":2,"syntology":null},{"url":"/paper/few-shot-object-detection-and-viewpoint","slug":"few-shot-object-detection-and-viewpoint","title":"Few-Shot Object Detection and Viewpoint Estimation for Objects in the Wild","date":"2020-07-23","arxiv_id":"2007.12107","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":14,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/few-shot-object-detection-and-viewpoint#ran","syntology_url":"https://syntology.ai/paper/2007.12107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12107"}},"official":null}},{"url":"/paper/pointcontrast-unsupervised-pre-training-for","slug":"pointcontrast-unsupervised-pre-training-for","title":"PointContrast: Unsupervised Pre-training for 3D Point Cloud Understanding","date":"2020-07-21","arxiv_id":"2007.10985","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pointcontrast-unsupervised-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2007.10985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.10985"}},"official":{"repos":["facebookresearch/PointContrast"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-and-reasoning-with-the-graph","slug":"learning-and-reasoning-with-the-graph","title":"Learning and Reasoning with the Graph Structure Representation in Robotic Surgery","date":"2020-07-07","arxiv_id":"2007.03357","repositories_listed":2,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-and-reasoning-with-the-graph#ran","syntology_url":"https://syntology.ai/paper/2007.03357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.03357"}},"official":{"repos":["mobarakol/Surgical_SceneGraph_Generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-visual-commonsense-for-robust-scene","slug":"learning-visual-commonsense-for-robust-scene","title":"Learning Visual Commonsense for Robust Scene Graph Generation","date":"2020-06-17","arxiv_id":"2006.09623","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-visual-commonsense-for-robust-scene#ran","syntology_url":"https://syntology.ai/paper/2006.09623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.09623"}},"official":null}},{"url":"/paper/self-supervised-scene-de-occlusion","slug":"self-supervised-scene-de-occlusion","title":"Self-Supervised Scene De-occlusion","date":"2020-04-06","arxiv_id":"2004.02788","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-scene-de-occlusion#ran","syntology_url":"https://syntology.ai/paper/2004.02788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.02788"}},"official":null}},{"url":"/paper/context-prior-for-scene-segmentation","slug":"context-prior-for-scene-segmentation","title":"Context Prior for Scene Segmentation","date":"2020-04-03","arxiv_id":"2004.01547","repositories_listed":2,"syntology":null},{"url":"/paper/image-segmentation-using-deep-learning-a","slug":"image-segmentation-using-deep-learning-a","title":"Image Segmentation Using Deep Learning: A Survey","date":"2020-01-15","arxiv_id":"2001.05566","repositories_listed":2,"syntology":null},{"url":"/paper/aerorit-a-new-scene-for-hyperspectral-image","slug":"aerorit-a-new-scene-for-hyperspectral-image","title":"AeroRIT: A New Scene for Hyperspectral Image Analysis","date":"2019-12-17","arxiv_id":"1912.08178","repositories_listed":2,"syntology":null},{"url":"/paper/cnn-based-lidar-point-cloud-de-noising-in","slug":"cnn-based-lidar-point-cloud-de-noising-in","title":"CNN-based Lidar Point Cloud De-Noising in Adverse Weather","date":"2019-12-09","arxiv_id":"1912.03874","repositories_listed":2,"syntology":null},{"url":"/paper/semantic-understanding-of-foggy-scenes-with","slug":"semantic-understanding-of-foggy-scenes-with","title":"Semantic Understanding of Foggy Scenes with Purely Synthetic Data","date":"2019-10-09","arxiv_id":"1910.03997","repositories_listed":2,"syntology":null},{"url":"/paper/global-aggregation-then-local-distribution-in","slug":"global-aggregation-then-local-distribution-in","title":"Global Aggregation then Local Distribution in Fully Convolutional Networks","date":"2019-09-16","arxiv_id":"1909.07229","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/global-aggregation-then-local-distribution-in#ran","syntology_url":"https://syntology.ai/paper/1909.07229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.07229"}},"official":{"repos":["lxtGH/GALD-Net"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gff-gated-fully-fusion-for-semantic","slug":"gff-gated-fully-fusion-for-semantic","title":"GFF: Gated Fully Fusion for Semantic Segmentation","date":"2019-04-03","arxiv_id":"1904.01803","repositories_listed":2,"syntology":null},{"url":"/paper/gated2depth-real-time-dense-lidar-from-gated","slug":"gated2depth-real-time-dense-lidar-from-gated","title":"Gated2Depth: Real-time Dense Lidar from Gated Images","date":"2019-02-13","arxiv_id":"1902.04997","repositories_listed":2,"syntology":null},{"url":"/paper/real-time-3d-traffic-cone-detection-for","slug":"real-time-3d-traffic-cone-detection-for","title":"Real-time 3D Traffic Cone Detection for Autonomous Driving","date":"2019-02-06","arxiv_id":"1902.02394","repositories_listed":2,"syntology":null},{"url":"/paper/skip-ganomaly-skip-connected-and","slug":"skip-ganomaly-skip-connected-and","title":"Skip-GANomaly: Skip Connected and Adversarially Trained Encoder-Decoder Anomaly Detection","date":"2019-01-25","arxiv_id":"1901.08954","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skip-ganomaly-skip-connected-and#ran","syntology_url":"https://syntology.ai/paper/1901.08954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.08954"}},"official":null}},{"url":"/paper/idd-a-dataset-for-exploring-problems-of","slug":"idd-a-dataset-for-exploring-problems-of","title":"IDD: A Dataset for Exploring Problems of Autonomous Navigation in Unconstrained Environments","date":"2018-11-26","arxiv_id":"1811.10200","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/idd-a-dataset-for-exploring-problems-of#ran","syntology_url":"https://syntology.ai/paper/1811.10200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10200"}},"official":null}},{"url":"/paper/visual-graphs-from-motion-vgfm-scene","slug":"visual-graphs-from-motion-vgfm-scene","title":"Visual Graphs from Motion (VGfM): Scene understanding with object geometry reasoning","date":"2018-07-16","arxiv_id":"1807.05933","repositories_listed":2,"syntology":null},{"url":"/paper/deepscores-a-dataset-for-segmentation","slug":"deepscores-a-dataset-for-segmentation","title":"DeepScores -- A Dataset for Segmentation, Detection and Classification of Tiny Objects","date":"2018-03-27","arxiv_id":"1804.00525","repositories_listed":2,"syntology":null},{"url":"/paper/blitznet-a-real-time-deep-network-for-scene","slug":"blitznet-a-real-time-deep-network-for-scene","title":"BlitzNet: A Real-Time Deep Network for Scene Understanding","date":"2017-08-09","arxiv_id":"1708.02813","repositories_listed":2,"syntology":null},{"url":"/paper/a-review-on-deep-learning-techniques-applied","slug":"a-review-on-deep-learning-techniques-applied","title":"A Review on Deep Learning Techniques Applied to Semantic Segmentation","date":"2017-04-22","arxiv_id":"1704.06857","repositories_listed":2,"syntology":null},{"url":"/paper/predicting-deeper-into-the-future-of-semantic","slug":"predicting-deeper-into-the-future-of-semantic","title":"Predicting Deeper into the Future of Semantic Segmentation","date":"2017-03-22","arxiv_id":"1703.07684","repositories_listed":2,"syntology":null},{"url":"/paper/visual-translation-embedding-network-for","slug":"visual-translation-embedding-network-for","title":"Visual Translation Embedding Network for Visual Relation Detection","date":"2017-02-27","arxiv_id":"1702.08319","repositories_listed":2,"syntology":null},{"url":"/paper/attend-infer-repeat-fast-scene-understanding","slug":"attend-infer-repeat-fast-scene-understanding","title":"Attend, Infer, Repeat: Fast Scene Understanding with Generative Models","date":"2016-03-28","arxiv_id":"1603.08575","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attend-infer-repeat-fast-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/1603.08575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.08575"}},"official":null}},{"url":"/paper/learning-to-tune-like-an-expert-interpretable","slug":"learning-to-tune-like-an-expert-interpretable","title":"Learning to Tune Like an Expert: Interpretable and Scene-Aware Navigation via MLLM Reasoning and CVAE-Based Adaptation","date":"2025-07-15","arxiv_id":"2507.11001","repositories_listed":1,"syntology":null},{"url":"/paper/ost-bench-evaluating-the-capabilities-of","slug":"ost-bench-evaluating-the-capabilities-of","title":"OST-Bench: Evaluating the Capabilities of MLLMs in Online Spatio-temporal Scene Understanding","date":"2025-07-10","arxiv_id":"2507.07984","repositories_listed":1,"syntology":null},{"url":"/paper/feed-forward-scenedino-for-unsupervised","slug":"feed-forward-scenedino-for-unsupervised","title":"Feed-Forward SceneDINO for Unsupervised Semantic Scene Completion","date":"2025-07-08","arxiv_id":"2507.06230","repositories_listed":1,"syntology":null},{"url":"/paper/siu3r-simultaneous-scene-understanding-and-3d","slug":"siu3r-simultaneous-scene-understanding-and-3d","title":"SIU3R: Simultaneous Scene Understanding and 3D Reconstruction Beyond Feature Alignment","date":"2025-07-03","arxiv_id":"2507.02705","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/siu3r-simultaneous-scene-understanding-and-3d#ran","syntology_url":"https://syntology.ai/paper/2507.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02705"}},"official":{"repos":["WU-CVGL/SIU3R"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surgtpgs-semantic-3d-surgical-scene","slug":"surgtpgs-semantic-3d-surgical-scene","title":"SurgTPGS: Semantic 3D Surgical Scene Understanding with Text Promptable Gaussian Splatting","date":"2025-06-29","arxiv_id":"2506.23309","repositories_listed":1,"syntology":null},{"url":"/paper/reme-a-data-centric-framework-for-training","slug":"reme-a-data-centric-framework-for-training","title":"ReME: A Data-Centric Framework for Training-Free Open-Vocabulary Segmentation","date":"2025-06-26","arxiv_id":"2506.21233","repositories_listed":1,"syntology":null},{"url":"/paper/dip-unsupervised-dense-in-context-post","slug":"dip-unsupervised-dense-in-context-post","title":"DIP: Unsupervised Dense In-Context Post-training of Visual Representations","date":"2025-06-23","arxiv_id":"2506.18463","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dip-unsupervised-dense-in-context-post#ran","syntology_url":"https://syntology.ai/paper/2506.18463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18463"}},"official":{"repos":["sirkosophia/dip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stsbench-a-spatio-temporal-scenario-benchmark","slug":"stsbench-a-spatio-temporal-scenario-benchmark","title":"STSBench: A Spatio-temporal Scenario Benchmark for Multi-modal Large Language Models in Autonomous Driving","date":"2025-06-06","arxiv_id":"2506.06218","repositories_listed":1,"syntology":null},{"url":"/paper/owmm-agent-open-world-mobile-manipulation","slug":"owmm-agent-open-world-mobile-manipulation","title":"OWMM-Agent: Open World Mobile Manipulation With Multi-modal Agentic Data Synthesis","date":"2025-06-04","arxiv_id":"2506.04217","repositories_listed":1,"syntology":null},{"url":"/paper/physgaia-a-physics-aware-dataset-of-multi","slug":"physgaia-a-physics-aware-dataset-of-multi","title":"PhysGaia: A Physics-Aware Dataset of Multi-Body Interactions for Dynamic Novel View Synthesis","date":"2025-06-03","arxiv_id":"2506.02794","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-prediction-meets-large-language","slug":"trajectory-prediction-meets-large-language","title":"Trajectory Prediction Meets Large Language Models: A Survey","date":"2025-06-03","arxiv_id":"2506.03408","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-videos-for-3d-world-enhancing","slug":"learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","arxiv_id":"2505.24625","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-from-videos-for-3d-world-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24625"}},"official":null}},{"url":"/paper/tackling-view-dependent-semantics-in-3d","slug":"tackling-view-dependent-semantics-in-3d","title":"Tackling View-Dependent Semantics in 3D Language Gaussian Splatting","date":"2025-05-30","arxiv_id":"2505.24746","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tackling-view-dependent-semantics-in-3d#ran","syntology_url":"https://syntology.ai/paper/2505.24746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24746"}},"official":{"repos":["sjtu-deepvisionlab/laga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conav-collaborative-cross-modal-reasoning-for","slug":"conav-collaborative-cross-modal-reasoning-for","title":"CoNav: Collaborative Cross-Modal Reasoning for Embodied Navigation","date":"2025-05-22","arxiv_id":"2505.16663","repositories_listed":1,"syntology":null},{"url":"/paper/dc-scene-data-centric-learning-for-3d-scene","slug":"dc-scene-data-centric-learning-for-3d-scene","title":"DC-Scene: Data-Centric Learning for 3D Scene Understanding","date":"2025-05-21","arxiv_id":"2505.15232","repositories_listed":1,"syntology":null},{"url":"/paper/apcotta-continual-test-time-adaptation-for","slug":"apcotta-continual-test-time-adaptation-for","title":"APCoTTA: Continual Test-Time Adaptation for Semantic Segmentation of Airborne LiDAR Point Clouds","date":"2025-05-15","arxiv_id":"2505.09971","repositories_listed":1,"syntology":null},{"url":"/paper/storyreasoning-dataset-using-chain-of-thought","slug":"storyreasoning-dataset-using-chain-of-thought","title":"StoryReasoning Dataset: Using Chain-of-Thought for Scene Understanding and Grounded Story Generation","date":"2025-05-15","arxiv_id":"2505.10292","repositories_listed":1,"syntology":null},{"url":"/paper/drrnet-macro-micro-feature-fusion-and-dual","slug":"drrnet-macro-micro-feature-fusion-and-dual","title":"DRRNet: Macro-Micro Feature Fusion and Dual Reverse Refinement for Camouflaged Object Detection","date":"2025-05-14","arxiv_id":"2505.09168","repositories_listed":1,"syntology":null},{"url":"/paper/extending-large-vision-language-model-for","slug":"extending-large-vision-language-model-for","title":"Extending Large Vision-Language Model for Diverse Interactive Tasks in Autonomous Driving","date":"2025-05-13","arxiv_id":"2505.08725","repositories_listed":1,"syntology":null},{"url":"/paper/hearing-and-seeing-through-clip-a-framework","slug":"hearing-and-seeing-through-clip-a-framework","title":"Hearing and Seeing Through CLIP: A Framework for Self-Supervised Sound Source Localization","date":"2025-05-08","arxiv_id":"2505.05343","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-feature-upsampling-methods-for","slug":"benchmarking-feature-upsampling-methods-for","title":"Benchmarking Feature Upsampling Methods for Vision Foundation Models using Interactive Segmentation","date":"2025-05-04","arxiv_id":"2505.02075","repositories_listed":1,"syntology":null},{"url":"/paper/llm-empowered-embodied-agent-for-memory","slug":"llm-empowered-embodied-agent-for-memory","title":"LLM-Empowered Embodied Agent for Memory-Augmented Task Planning in Household Robotics","date":"2025-04-30","arxiv_id":"2504.21716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-empowered-embodied-agent-for-memory#ran","syntology_url":"https://syntology.ai/paper/2504.21716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21716"}},"official":{"repos":["marc1198/chat-hsr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-vision-llms-road-ready-a-comprehensive","slug":"are-vision-llms-road-ready-a-comprehensive","title":"Are Vision LLMs Road-Ready? A Comprehensive Benchmark for Safety-Critical Driving Video Understanding","date":"2025-04-20","arxiv_id":"2504.14526","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-automatic-cad-annotations-for","slug":"leveraging-automatic-cad-annotations-for","title":"Leveraging Automatic CAD Annotations for Supervised Learning in 3D Scene Understanding","date":"2025-04-18","arxiv_id":"2504.13580","repositories_listed":1,"syntology":null},{"url":"/paper/training-free-hierarchical-scene","slug":"training-free-hierarchical-scene","title":"Training-Free Hierarchical Scene Understanding for Gaussian Splatting with Superpoint Graphs","date":"2025-04-17","arxiv_id":"2504.13153","repositories_listed":1,"syntology":null},{"url":"/paper/dc-sam-in-context-segment-anything-in-images","slug":"dc-sam-in-context-segment-anything-in-images","title":"DC-SAM: In-Context Segment Anything in Images and Videos via Dual Consistency","date":"2025-04-16","arxiv_id":"2504.12080","repositories_listed":1,"syntology":null},{"url":"/paper/soccernet-v3d-leveraging-sports-broadcast","slug":"soccernet-v3d-leveraging-sports-broadcast","title":"SoccerNet-v3D: Leveraging Sports Broadcast Replays for 3D Scene Understanding","date":"2025-04-14","arxiv_id":"2504.10106","repositories_listed":1,"syntology":null},{"url":"/paper/masked-scene-modeling-narrowing-the-gap","slug":"masked-scene-modeling-narrowing-the-gap","title":"Masked Scene Modeling: Narrowing the Gap Between Supervised and Self-Supervised Learning in 3D Scene Understanding","date":"2025-04-09","arxiv_id":"2504.06719","repositories_listed":1,"syntology":null},{"url":"/paper/camcontexti2v-context-aware-controllable","slug":"camcontexti2v-context-aware-controllable","title":"CamContextI2V: Context-aware Controllable Video Generation","date":"2025-04-08","arxiv_id":"2504.06022","repositories_listed":1,"syntology":null},{"url":"/paper/dformerv2-geometry-self-attention-for-rgbd","slug":"dformerv2-geometry-self-attention-for-rgbd","title":"DFormerv2: Geometry Self-Attention for RGBD Semantic Segmentation","date":"2025-04-07","arxiv_id":"2504.04701","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/dformerv2-geometry-self-attention-for-rgbd#ran","syntology_url":"https://syntology.ai/paper/2504.04701","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04701"}},"official":{"repos":["VCIP-RGBD/DFormer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/planning-safety-trajectories-with-dual-phase","slug":"planning-safety-trajectories-with-dual-phase","title":"Planning Safety Trajectories with Dual-Phase, Physics-Informed, and Transportation Knowledge-Driven Large Language Models","date":"2025-04-06","arxiv_id":"2504.04562","repositories_listed":1,"syntology":null},{"url":"/paper/f-vita-foundation-model-guided-visible-to","slug":"f-vita-foundation-model-guided-visible-to","title":"F-ViTA: Foundation Model Guided Visible to Thermal Translation","date":"2025-04-03","arxiv_id":"2504.02801","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-fusion-and-vision-language-models","slug":"multimodal-fusion-and-vision-language-models","title":"Multimodal Fusion and Vision-Language Models: A Survey for Robot Vision","date":"2025-04-03","arxiv_id":"2504.02477","repositories_listed":1,"syntology":null},{"url":"/paper/scene-centric-unsupervised-panoptic","slug":"scene-centric-unsupervised-panoptic","title":"Scene-Centric Unsupervised Panoptic Segmentation","date":"2025-04-02","arxiv_id":"2504.01955","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/scene-centric-unsupervised-panoptic#ran","syntology_url":"https://syntology.ai/paper/2504.01955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01955"}},"official":{"repos":["visinf/cups"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/wikivideo-article-generation-from-multiple","slug":"wikivideo-article-generation-from-multiple","title":"WikiVideo: Article Generation from Multiple Videos","date":"2025-04-01","arxiv_id":"2504.00939","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-omnidirectional-stereo-matching-with","slug":"boosting-omnidirectional-stereo-matching-with","title":"Boosting Omnidirectional Stereo Matching with a Pre-trained Depth Foundation Model","date":"2025-03-30","arxiv_id":"2503.23502","repositories_listed":1,"syntology":null},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","slug":"opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/opendrivevla-towards-end-to-end-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.23463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23463"}},"official":{"repos":["DriveVLA/OpenDriveVLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-compositional-scene-understanding","slug":"evaluating-compositional-scene-understanding","title":"Evaluating Compositional Scene Understanding in Multimodal Generative Models","date":"2025-03-29","arxiv_id":"2503.23125","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-trade-off-stream-and-query-guided","slug":"mitigating-trade-off-stream-and-query-guided","title":"Mitigating Trade-off: Stream and Query-guided Aggregation for Efficient and Effective 3D Occupancy Prediction","date":"2025-03-28","arxiv_id":"2503.22087","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generating-realistic-3d-semantic","slug":"towards-generating-realistic-3d-semantic","title":"Towards Generating Realistic 3D Semantic Training Data for Autonomous Driving","date":"2025-03-27","arxiv_id":"2503.21449","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generating-realistic-3d-semantic#ran","syntology_url":"https://syntology.ai/paper/2503.21449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21449"}},"official":{"repos":["prbonn/3diss"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cob-gs-clear-object-boundaries-in-3dgs","slug":"cob-gs-clear-object-boundaries-in-3dgs","title":"COB-GS: Clear Object Boundaries in 3DGS Segmentation Based on Boundary-Adaptive Gaussian Splitting","date":"2025-03-25","arxiv_id":"2503.19443","repositories_listed":1,"syntology":null},{"url":"/paper/superflow-enhanced-spatiotemporal-consistency","slug":"superflow-enhanced-spatiotemporal-consistency","title":"SuperFlow++: Enhanced Spatiotemporal Consistency for Cross-Modal Data Pretraining","date":"2025-03-25","arxiv_id":"2503.19912","repositories_listed":1,"syntology":null},{"url":"/paper/the-coralscapes-dataset-semantic-scene","slug":"the-coralscapes-dataset-semantic-scene","title":"The Coralscapes Dataset: Semantic Scene Understanding in Coral Reefs","date":"2025-03-25","arxiv_id":"2503.20000","repositories_listed":1,"syntology":null},{"url":"/paper/polarfree-polarization-based-reflection-free","slug":"polarfree-polarization-based-reflection-free","title":"PolarFree: Polarization-based Reflection-free Imaging","date":"2025-03-23","arxiv_id":"2503.18055","repositories_listed":1,"syntology":null},{"url":"/paper/scenesplat-gaussian-splatting-based-scene","slug":"scenesplat-gaussian-splatting-based-scene","title":"SceneSplat: Gaussian Splatting-based Scene Understanding with Vision-Language Pretraining","date":"2025-03-23","arxiv_id":"2503.18052","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-and-uncertainty-aware","slug":"cross-modal-and-uncertainty-aware","title":"Cross-Modal and Uncertainty-Aware Agglomeration for Open-Vocabulary 3D Scene Understanding","date":"2025-03-20","arxiv_id":"2503.16707","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-and-uncertainty-aware#ran","syntology_url":"https://syntology.ai/paper/2503.16707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16707"}},"official":{"repos":["tyroneli/cua_o3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crab-a-unified-audio-visual-scene-1","slug":"crab-a-unified-audio-visual-scene-1","title":"Crab: A Unified Audio-Visual Scene Understanding Model with Explicit Cooperation","date":"2025-03-17","arxiv_id":"2503.13068","repositories_listed":1,"syntology":null},{"url":"/paper/nuplanqa-a-large-scale-dataset-and-benchmark","slug":"nuplanqa-a-large-scale-dataset-and-benchmark","title":"NuPlanQA: A Large-Scale Dataset and Benchmark for Multi-View Driving Scene Understanding in Multi-Modal Large Language Models","date":"2025-03-17","arxiv_id":"2503.12772","repositories_listed":1,"syntology":null},{"url":"/paper/logic-rag-augmenting-large-multimodal-models","slug":"logic-rag-augmenting-large-multimodal-models","title":"Logic-RAG: Augmenting Large Multimodal Models with Visual-Spatial Knowledge for Road Scene Understanding","date":"2025-03-16","arxiv_id":"2503.12663","repositories_listed":1,"syntology":null},{"url":"/paper/trackocc-camera-based-4d-panoptic-occupancy","slug":"trackocc-camera-based-4d-panoptic-occupancy","title":"TrackOcc: Camera-based 4D Panoptic Occupancy Tracking","date":"2025-03-11","arxiv_id":"2503.08471","repositories_listed":1,"syntology":null},{"url":"/paper/a-data-centric-revisit-of-pre-trained-vision","slug":"a-data-centric-revisit-of-pre-trained-vision","title":"A Data-Centric Revisit of Pre-Trained Vision Models for Robot Learning","date":"2025-03-10","arxiv_id":"2503.06960","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-data-centric-revisit-of-pre-trained-vision#ran","syntology_url":"https://syntology.ai/paper/2503.06960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06960"}},"official":{"repos":["cvmi-lab/slotmim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chameleon-fast-slow-neuro-symbolic-lane","slug":"chameleon-fast-slow-neuro-symbolic-lane","title":"Chameleon: Fast-slow Neuro-symbolic Lane Topology Extraction","date":"2025-03-10","arxiv_id":"2503.07485","repositories_listed":1,"syntology":null},{"url":"/paper/towards-ambiguity-free-spatial-foundation","slug":"towards-ambiguity-free-spatial-foundation","title":"Towards Ambiguity-Free Spatial Foundation Model: Rethinking and Decoupling Depth Ambiguity","date":"2025-03-08","arxiv_id":"2503.06014","repositories_listed":1,"syntology":null},{"url":"/paper/vlscene-vision-language-guidance-distillation","slug":"vlscene-vision-language-guidance-distillation","title":"VLScene: Vision-Language Guidance Distillation for Camera-Based 3D Semantic Scene Completion","date":"2025-03-08","arxiv_id":"2503.06219","repositories_listed":1,"syntology":null},{"url":"/paper/an-egocentric-vision-language-model-based","slug":"an-egocentric-vision-language-model-based","title":"An Egocentric Vision-Language Model based Portable Real-time Smart Assistant","date":"2025-03-06","arxiv_id":"2503.04250","repositories_listed":1,"syntology":null},{"url":"/paper/inst3d-lmm-instance-aware-3d-scene","slug":"inst3d-lmm-instance-aware-3d-scene","title":"Inst3D-LMM: Instance-Aware 3D Scene Understanding with Multi-modal Instruction Tuning","date":"2025-03-01","arxiv_id":"2503.00513","repositories_listed":1,"syntology":null},{"url":"/paper/distill-any-depth-distillation-creates-a","slug":"distill-any-depth-distillation-creates-a","title":"Distill Any Depth: Distillation Creates a Stronger Monocular Depth Estimator","date":"2025-02-26","arxiv_id":"2502.19204","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-context-transformer-for-multi","slug":"hierarchical-context-transformer-for-multi","title":"Hierarchical Context Transformer for Multi-level Semantic Scene Understanding","date":"2025-02-21","arxiv_id":"2502.15184","repositories_listed":1,"syntology":null},{"url":"/paper/crossover-3d-scene-cross-modal-alignment","slug":"crossover-3d-scene-cross-modal-alignment","title":"CrossOver: 3D Scene Cross-Modal Alignment","date":"2025-02-20","arxiv_id":"2502.15011","repositories_listed":1,"syntology":null},{"url":"/paper/navrag-generating-user-demand-instructions","slug":"navrag-generating-user-demand-instructions","title":"NavRAG: Generating User Demand Instructions for Embodied Navigation through Retrieval-Augmented LLM","date":"2025-02-16","arxiv_id":"2502.11142","repositories_listed":1,"syntology":null},{"url":"/paper/occlusion-aware-non-rigid-point-cloud","slug":"occlusion-aware-non-rigid-point-cloud","title":"Occlusion-aware Non-Rigid Point Cloud Registration via Unsupervised Neural Deformation Correntropy","date":"2025-02-15","arxiv_id":"2502.10704","repositories_listed":1,"syntology":null},{"url":"/paper/event-aided-semantic-scene-completion","slug":"event-aided-semantic-scene-completion","title":"Event-aided Semantic Scene Completion","date":"2025-02-04","arxiv_id":"2502.02334","repositories_listed":1,"syntology":null},{"url":"/paper/hermes-a-unified-self-driving-world-model-for","slug":"hermes-a-unified-self-driving-world-model-for","title":"HERMES: A Unified Self-Driving World Model for Simultaneous 3D Scene Understanding and Generation","date":"2025-01-24","arxiv_id":"2501.14729","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hermes-a-unified-self-driving-world-model-for#ran","syntology_url":"https://syntology.ai/paper/2501.14729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14729"}},"official":{"repos":["lmd0311/hermes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/endochat-grounded-multimodal-large-language","slug":"endochat-grounded-multimodal-large-language","title":"EndoChat: Grounded Multimodal Large Language Model for Endoscopic Surgery","date":"2025-01-20","arxiv_id":"2501.11347","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/endochat-grounded-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2501.11347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11347"}},"official":{"repos":["gkw0010/endochat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3ur-llm-an-end-to-end-multimodal-large","slug":"3ur-llm-an-end-to-end-multimodal-large","title":"3UR-LLM: An End-to-End Multimodal Large Language Model for 3D Scene Understanding","date":"2025-01-14","arxiv_id":"2501.07819","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-superpixel-segmentation-via","slug":"hierarchical-superpixel-segmentation-via","title":"Hierarchical Superpixel Segmentation via Structural Information Theory","date":"2025-01-13","arxiv_id":"2501.07069","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-partial-cycle-consistency-for","slug":"self-supervised-partial-cycle-consistency-for","title":"Self-Supervised Partial Cycle-Consistency for Multi-View Matching","date":"2025-01-10","arxiv_id":"2501.06000","repositories_listed":1,"syntology":null},{"url":"/paper/videolifter-lifting-videos-to-3d-with-fast","slug":"videolifter-lifting-videos-to-3d-with-fast","title":"VideoLifter: Lifting Videos to 3D with Fast Hierarchical Stereo Alignment","date":"2025-01-03","arxiv_id":"2501.01949","repositories_listed":1,"syntology":null},{"url":"/paper/gpt4scene-understand-3d-scenes-from-videos","slug":"gpt4scene-understand-3d-scenes-from-videos","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","date":"2025-01-02","arxiv_id":"2501.01428","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt4scene-understand-3d-scenes-from-videos#ran","syntology_url":"https://syntology.ai/paper/2501.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01428"}},"official":{"repos":["Qi-Zhangyang/GPT4Scene"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/all-day-multi-camera-multi-target-tracking","slug":"all-day-multi-camera-multi-target-tracking","title":"All-Day Multi-Camera Multi-Target Tracking","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/avqacl-a-novel-benchmark-for-audio-visual","slug":"avqacl-a-novel-benchmark-for-audio-visual","title":"AVQACL: A Novel Benchmark for Audio-Visual Question Answering Continual Learning","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/object-aware-sound-source-localization-via","slug":"object-aware-sound-source-localization-via","title":"Object-aware Sound Source Localization via Audio-Visual Scene Understanding","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/storm-spatio-temporal-reconstruction-model","slug":"storm-spatio-temporal-reconstruction-model","title":"STORM: Spatio-Temporal Reconstruction Model for Large-Scale Outdoor Scenes","date":"2024-12-31","arxiv_id":"2501.00602","repositories_listed":1,"syntology":null},{"url":"/paper/mllm-sul-multimodal-large-language-model-for","slug":"mllm-sul-multimodal-large-language-model-for","title":"MLLM-SUL: Multimodal Large Language Model for Semantic Scene Understanding and Localization in Traffic Scenarios","date":"2024-12-27","arxiv_id":"2412.19406","repositories_listed":1,"syntology":null}],"record_sha256":"18ae69d132e18df712b864eed57179140651f6565954ff14d09ee3f82b8deca0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}