{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/3","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":18,"rows_per_page":100,"rows":[201,300],"of":1723,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding","prev":"/task/scene-understanding/papers/2","next":"/task/scene-understanding/papers/4","papers":[{"url":"/paper/3dgraphllm-combining-semantic-graphs-and","slug":"3dgraphllm-combining-semantic-graphs-and","title":"3DGraphLLM: Combining Semantic Graphs and Large Language Models for 3D Scene Understanding","date":"2024-12-24","arxiv_id":"2412.18450","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3dgraphllm-combining-semantic-graphs-and#ran","syntology_url":"https://syntology.ai/paper/2412.18450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18450"}},"official":{"repos":["cognitiveaisystems/3dgraphllm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parallel-neural-computing-for-scene","slug":"parallel-neural-computing-for-scene","title":"Parallel Neural Computing for Scene Understanding from LiDAR Perception in Autonomous Racing","date":"2024-12-24","arxiv_id":"2412.18165","repositories_listed":1,"syntology":null},{"url":"/paper/improving-object-detection-for-time-lapse","slug":"improving-object-detection-for-time-lapse","title":"Improving Object Detection for Time-Lapse Imagery Using Temporal Features in Wildlife Monitoring","date":"2024-12-20","arxiv_id":"2412.16329","repositories_listed":1,"syntology":null},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pc-bev-an-efficient-polar-cartesian-bev","slug":"pc-bev-an-efficient-polar-cartesian-bev","title":"PC-BEV: An Efficient Polar-Cartesian BEV Fusion Framework for LiDAR Semantic Segmentation","date":"2024-12-19","arxiv_id":"2412.14821","repositories_listed":1,"syntology":null},{"url":"/paper/relationfield-relate-anything-in-radiance","slug":"relationfield-relate-anything-in-radiance","title":"RelationField: Relate Anything in Radiance Fields","date":"2024-12-18","arxiv_id":"2412.13652","repositories_listed":1,"syntology":null},{"url":"/paper/dense-audio-visual-event-localization-under","slug":"dense-audio-visual-event-localization-under","title":"Dense Audio-Visual Event Localization under Cross-Modal Consistency and Multi-Temporal Granularity Collaboration","date":"2024-12-17","arxiv_id":"2412.12628","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dense-audio-visual-event-localization-under#ran","syntology_url":"https://syntology.ai/paper/2412.12628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12628"}},"official":{"repos":["zzhhfut/ccnet-aaai2025"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emma-x-an-embodied-multimodal-action-model","slug":"emma-x-an-embodied-multimodal-action-model","title":"Emma-X: An Embodied Multimodal Action Model with Grounded Chain of Thought and Look-ahead Spatial Reasoning","date":"2024-12-16","arxiv_id":"2412.11974","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-dino-foresight-looking-into-the-future","slug":"texttt-dino-foresight-looking-into-the-future","title":"$\\texttt{DINO-Foresight}$: Looking into the Future with DINO","date":"2024-12-16","arxiv_id":"2412.11673","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/texttt-dino-foresight-looking-into-the-future#ran","syntology_url":"https://syntology.ai/paper/2412.11673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11673"}},"official":{"repos":["sta8is/dino-foresight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wisead-knowledge-augmented-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.09951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09951"}},"official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-spacesgg-visual-instruct-tuning-for","slug":"llava-spacesgg-visual-instruct-tuning-for","title":"LLaVA-SpaceSGG: Visual Instruct Tuning for Open-vocabulary Scene Graph Generation with Enhanced Spatial Relations","date":"2024-12-09","arxiv_id":"2412.06322","repositories_listed":1,"syntology":null},{"url":"/paper/stag-1-towards-realistic-4d-driving","slug":"stag-1-towards-realistic-4d-driving","title":"Stag-1: Towards Realistic 4D Driving Simulation with Video Generation Model","date":"2024-12-06","arxiv_id":"2412.05280","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stag-1-towards-realistic-4d-driving#ran","syntology_url":"https://syntology.ai/paper/2412.05280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05280"}},"official":{"repos":["wzzheng/stag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lscenellm-enhancing-large-3d-scene","slug":"lscenellm-enhancing-large-3d-scene","title":"LSceneLLM: Enhancing Large 3D Scene Understanding Using Adaptive Visual Preferences","date":"2024-12-02","arxiv_id":"2412.01292","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lscenellm-enhancing-large-3d-scene#ran","syntology_url":"https://syntology.ai/paper/2412.01292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01292"}},"official":{"repos":["Hoyyyaard/LSceneLLM"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/video-3d-llm-learning-position-aware-video","slug":"video-3d-llm-learning-position-aware-video","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","date":"2024-11-30","arxiv_id":"2412.00493","repositories_listed":1,"syntology":null},{"url":"/paper/bootstraping-clustering-of-gaussians-for-view","slug":"bootstraping-clustering-of-gaussians-for-view","title":"Bootstraping Clustering of Gaussians for View-consistent 3D Scene Understanding","date":"2024-11-29","arxiv_id":"2411.19551","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bootstraping-clustering-of-gaussians-for-view#ran","syntology_url":"https://syntology.ai/paper/2411.19551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19551"}},"official":{"repos":["wb014/FreeGS"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geobench-vlm-benchmarking-vision-language","slug":"geobench-vlm-benchmarking-vision-language","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","date":"2024-11-28","arxiv_id":"2411.19325","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/geobench-vlm-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.19325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19325"}},"official":{"repos":["the-ai-alliance/geo-bench-vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grid-augumented-vision-a-simple-yet-effective","slug":"grid-augumented-vision-a-simple-yet-effective","title":"Grid-augmented vision: A simple yet effective approach for enhanced spatial understanding in multi-modal agents","date":"2024-11-27","arxiv_id":"2411.18270","repositories_listed":1,"syntology":null},{"url":"/paper/locate-gat-modeling-multi-scale-local-context","slug":"locate-gat-modeling-multi-scale-local-context","title":"LoCATe-GAT: Modeling Multi-Scale Local Context and Action Relationships for Zero-Shot Action Recognition","date":"2024-11-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/box-for-mask-and-mask-for-box-weak-losses-for","slug":"box-for-mask-and-mask-for-box-weak-losses-for","title":"Box for Mask and Mask for Box: weak losses for multi-task partially supervised learning","date":"2024-11-26","arxiv_id":"2411.17536","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-robust-point-cloud-semantic","slug":"an-end-to-end-robust-point-cloud-semantic","title":"An End-to-End Robust Point Cloud Semantic Segmentation Network with Single-Step Conditional Diffusion Models","date":"2024-11-25","arxiv_id":"2411.16308","repositories_listed":1,"syntology":null},{"url":"/paper/gaussianpretrain-a-simple-unified-3d-gaussian","slug":"gaussianpretrain-a-simple-unified-3d-gaussian","title":"GaussianPretrain: A Simple Unified 3D Gaussian Representation for Visual Pre-training in Autonomous Driving","date":"2024-11-19","arxiv_id":"2411.12452","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gaussianpretrain-a-simple-unified-3d-gaussian#ran","syntology_url":"https://syntology.ai/paper/2411.12452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12452"}},"official":{"repos":["public-bots/gaussianpretrain"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metricgold-leveraging-text-to-image-latent","slug":"metricgold-leveraging-text-to-image-latent","title":"MetricGold: Leveraging Text-To-Image Latent Diffusion Models for Metric Depth Estimation","date":"2024-11-16","arxiv_id":"2411.10886","repositories_listed":1,"syntology":null},{"url":"/paper/tesgnn-temporal-equivariant-scene-graph","slug":"tesgnn-temporal-equivariant-scene-graph","title":"TESGNN: Temporal Equivariant Scene Graph Neural Networks for Efficient and Robust Multi-View 3D Scene Understanding","date":"2024-11-15","arxiv_id":"2411.10509","repositories_listed":1,"syntology":null},{"url":"/paper/osmloc-single-image-based-visual-localization","slug":"osmloc-single-image-based-visual-localization","title":"OSMLoc: Single Image-Based Visual Localization in OpenStreetMap with Fused Geometric and Semantic Guidance","date":"2024-11-13","arxiv_id":"2411.08665","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-geometric-estimation-of-depth-and","slug":"multi-task-geometric-estimation-of-depth-and","title":"Multi-task Geometric Estimation of Depth and Surface Normal from Monocular 360° Images","date":"2024-11-04","arxiv_id":"2411.01749","repositories_listed":1,"syntology":null},{"url":"/paper/senna-bridging-large-vision-language-models","slug":"senna-bridging-large-vision-language-models","title":"Senna: Bridging Large Vision-Language Models and End-to-End Autonomous Driving","date":"2024-10-29","arxiv_id":"2410.22313","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/senna-bridging-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.22313","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22313"}},"official":{"repos":["hustvl/senna"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-scene-segmentation-by-transformer","slug":"surgical-scene-segmentation-by-transformer","title":"Surgical Scene Segmentation by Transformer With Asymmetric Feature Enhancement","date":"2024-10-23","arxiv_id":"2410.17642","repositories_listed":1,"syntology":null},{"url":"/paper/part-whole-relational-fusion-towards-multi","slug":"part-whole-relational-fusion-towards-multi","title":"Part-Whole Relational Fusion Towards Multi-Modal Scene Understanding","date":"2024-10-19","arxiv_id":"2410.14944","repositories_listed":1,"syntology":null},{"url":"/paper/arkit-labelmaker-a-new-scale-for-indoor-3d","slug":"arkit-labelmaker-a-new-scale-for-indoor-3d","title":"ARKit LabelMaker: A New Scale for Indoor 3D Scene Understanding","date":"2024-10-17","arxiv_id":"2410.13924","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/arkit-labelmaker-a-new-scale-for-indoor-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13924","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13924"}},"official":{"repos":["cvg/labelmaker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d","slug":"vlm-grounder-a-vlm-agent-for-zero-shot-3d","title":"VLM-Grounder: A VLM Agent for Zero-Shot 3D Visual Grounding","date":"2024-10-17","arxiv_id":"2410.13860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13860"}},"official":{"repos":["openrobotlab/vlm-grounder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/loli-street-benchmarking-low-light-image","slug":"loli-street-benchmarking-low-light-image","title":"LoLI-Street: Benchmarking Low-Light Image Enhancement and Beyond","date":"2024-10-13","arxiv_id":"2410.09831","repositories_listed":1,"syntology":null},{"url":"/paper/resscal3d-joint-acquisition-and-semantic","slug":"resscal3d-joint-acquisition-and-semantic","title":"RESSCAL3D++: Joint Acquisition and Semantic Segmentation of 3D Point Clouds","date":"2024-10-03","arxiv_id":"2410.02323","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-assessment-of-visual-sound-source","slug":"a-critical-assessment-of-visual-sound-source","title":"A Critical Assessment of Visual Sound Source Localization Models Including Negative Audio","date":"2024-10-01","arxiv_id":"2410.01020","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multiple-probabilistic-decisions","slug":"learning-multiple-probabilistic-decisions","title":"Learning Multiple Probabilistic Decisions from Latent World Model in Autonomous Driving","date":"2024-09-24","arxiv_id":"2409.15730","repositories_listed":1,"syntology":null},{"url":"/paper/clair-a-leveraging-large-language-models-to","slug":"clair-a-leveraging-large-language-models-to","title":"CLAIR-A: Leveraging Large Language Models to Judge Audio Captions","date":"2024-09-19","arxiv_id":"2409.12962","repositories_listed":1,"syntology":null},{"url":"/paper/hi-slam-scaling-up-semantics-in-slam-with-a","slug":"hi-slam-scaling-up-semantics-in-slam-with-a","title":"Hier-SLAM: Scaling-up Semantics in SLAM with a Hierarchically Categorical Gaussian Splatting","date":"2024-09-19","arxiv_id":"2409.12518","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hi-slam-scaling-up-semantics-in-slam-with-a#ran","syntology_url":"https://syntology.ai/paper/2409.12518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12518"}},"official":{"repos":["LeeBY68/Hier-SLAM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/daf-net-a-dual-branch-feature-decomposition","slug":"daf-net-a-dual-branch-feature-decomposition","title":"DAF-Net: A Dual-Branch Feature Decomposition Fusion Network with Domain Adaptive for Infrared and Visible Image Fusion","date":"2024-09-18","arxiv_id":"2409.11642","repositories_listed":1,"syntology":null},{"url":"/paper/towards-global-localization-using-multi-modal","slug":"towards-global-localization-using-multi-modal","title":"Towards Global Localization using Multi-Modal Object-Instance Re-Identification","date":"2024-09-18","arxiv_id":"2409.12002","repositories_listed":1,"syntology":null},{"url":"/paper/primedepth-efficient-monocular-depth","slug":"primedepth-efficient-monocular-depth","title":"PrimeDepth: Efficient Monocular Depth Estimation with a Stable Diffusion Preimage","date":"2024-09-13","arxiv_id":"2409.09144","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":3,"n_no_contract":5,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 3 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/primedepth-efficient-monocular-depth#ran","syntology_url":"https://syntology.ai/paper/2409.09144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.09144"}},"official":{"repos":["vislearn/PrimeDepth"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/led-light-enhanced-depth-estimation-at-night","slug":"led-light-enhanced-depth-estimation-at-night","title":"LED: Light Enhanced Depth Estimation at Night","date":"2024-09-12","arxiv_id":"2409.08031","repositories_listed":1,"syntology":null},{"url":"/paper/loss-distillation-via-gradient-matching-for","slug":"loss-distillation-via-gradient-matching-for","title":"Loss Distillation via Gradient Matching for Point Cloud Completion with Weighted Chamfer Distance","date":"2024-09-10","arxiv_id":"2409.06171","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/loss-distillation-via-gradient-matching-for#ran","syntology_url":"https://syntology.ai/paper/2409.06171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06171"}},"official":{"repos":["zhang-vislab/iros2024-lossdistillationweightedcd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/online-3d-reconstruction-and-dense-tracking","slug":"online-3d-reconstruction-and-dense-tracking","title":"Online 3D reconstruction and dense tracking in endoscopic videos","date":"2024-09-09","arxiv_id":"2409.06037","repositories_listed":1,"syntology":null},{"url":"/paper/rcnet-deep-recurrent-collaborative-network","slug":"rcnet-deep-recurrent-collaborative-network","title":"RCNet: Deep Recurrent Collaborative Network for Multi-View Low-Light Image Enhancement","date":"2024-09-06","arxiv_id":"2409.04363","repositories_listed":1,"syntology":null},{"url":"/paper/lexicon3d-probing-visual-foundation-models","slug":"lexicon3d-probing-visual-foundation-models","title":"Lexicon3D: Probing Visual Foundation Models for Complex 3D Scene Understanding","date":"2024-09-05","arxiv_id":"2409.03757","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lexicon3d-probing-visual-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2409.03757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03757"}},"official":null}},{"url":"/paper/eprecon-an-efficient-framework-for-real-time","slug":"eprecon-an-efficient-framework-for-real-time","title":"EPRecon: An Efficient Framework for Real-Time Panoptic 3D Reconstruction from Monocular Video","date":"2024-09-03","arxiv_id":"2409.01807","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-deep-shadows-a-survey-on-image-and","slug":"unveiling-deep-shadows-a-survey-on-image-and","title":"Unveiling Deep Shadows: A Survey and Benchmark on Image and Video Shadow Detection, Removal, and Generation in the Deep Learning Era","date":"2024-09-03","arxiv_id":"2409.02108","repositories_listed":1,"syntology":null},{"url":"/paper/adaptvision-dynamic-input-scaling-in-mllms","slug":"adaptvision-dynamic-input-scaling-in-mllms","title":"AdaptVision: Dynamic Input Scaling in MLLMs for Versatile Scene Understanding","date":"2024-08-30","arxiv_id":"2408.16986","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptvision-dynamic-input-scaling-in-mllms#ran","syntology_url":"https://syntology.ai/paper/2408.16986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16986"}},"official":{"repos":["harrytea/adaptvision"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/urbench-a-comprehensive-benchmark-for","slug":"urbench-a-comprehensive-benchmark-for","title":"UrBench: A Comprehensive Benchmark for Evaluating Large Multimodal Models in Multi-View Urban Scenarios","date":"2024-08-30","arxiv_id":"2408.17267","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/urbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2408.17267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.17267"}},"official":{"repos":["opendatalab/urbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robosense-large-scale-dataset-and-benchmark","slug":"robosense-large-scale-dataset-and-benchmark","title":"RoboSense: Large-scale Dataset and Benchmark for Egocentric Robot Perception and Navigation in Crowded and Unstructured Environments","date":"2024-08-28","arxiv_id":"2408.15503","repositories_listed":1,"syntology":null},{"url":"/paper/handling-geometric-domain-shifts-in-semantic","slug":"handling-geometric-domain-shifts-in-semantic","title":"Handling Geometric Domain Shifts in Semantic Segmentation of Surgical RGB and Hyperspectral Images","date":"2024-08-27","arxiv_id":"2408.15373","repositories_listed":1,"syntology":null},{"url":"/paper/mtmamba-enhancing-multi-task-dense-scene-1","slug":"mtmamba-enhancing-multi-task-dense-scene-1","title":"MTMamba++: Enhancing Multi-Task Dense Scene Understanding via Mamba-Based Decoders","date":"2024-08-27","arxiv_id":"2408.15101","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtmamba-enhancing-multi-task-dense-scene-1#ran","syntology_url":"https://syntology.ai/paper/2408.15101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15101"}},"official":{"repos":["envision-research/mtmamba"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rsteller-scaling-up-visual-language-modeling","slug":"rsteller-scaling-up-visual-language-modeling","title":"RSTeller: Scaling Up Visual Language Modeling in Remote Sensing with Rich Linguistic Semantics from Openly Available Data and Large Language Models","date":"2024-08-27","arxiv_id":"2408.14744","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rsteller-scaling-up-visual-language-modeling#ran","syntology_url":"https://syntology.ai/paper/2408.14744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14744"}},"official":{"repos":["slytheringe/rsteller"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extremely-fine-grained-visual-classification","slug":"extremely-fine-grained-visual-classification","title":"Extremely Fine-Grained Visual Classification over Resembling Glyphs in the Wild","date":"2024-08-25","arxiv_id":"2408.13774","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-scene-coherence-for-semi-supervised","slug":"exploring-scene-coherence-for-semi-supervised","title":"Exploring Scene Affinity for Semi-Supervised LiDAR Semantic Segmentation","date":"2024-08-21","arxiv_id":"2408.11280","repositories_listed":1,"syntology":null},{"url":"/paper/openscan-a-benchmark-for-generalized-open","slug":"openscan-a-benchmark-for-generalized-open","title":"OpenScan: A Benchmark for Generalized Open-Vocabulary 3D Scene Understanding","date":"2024-08-20","arxiv_id":"2408.11030","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openscan-a-benchmark-for-generalized-open#ran","syntology_url":"https://syntology.ai/paper/2408.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11030"}},"official":{"repos":["youjunzhao/openscan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepinteraction-multi-modality-interaction","slug":"deepinteraction-multi-modality-interaction","title":"DeepInteraction++: Multi-Modality Interaction for Autonomous Driving","date":"2024-08-09","arxiv_id":"2408.05075","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepinteraction-multi-modality-interaction#ran","syntology_url":"https://syntology.ai/paper/2408.05075","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.05075"}},"official":{"repos":["fudan-zvg/deepinteraction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-llms-for-enhanced-open-vocabulary","slug":"leveraging-llms-for-enhanced-open-vocabulary","title":"Query3D: LLM-Powered Open-Vocabulary Scene Segmentation with Language Embedded 3D Gaussian","date":"2024-08-07","arxiv_id":"2408.03516","repositories_listed":1,"syntology":null},{"url":"/paper/a-plug-and-play-method-for-rare-human-object","slug":"a-plug-and-play-method-for-rare-human-object","title":"A Plug-and-Play Method for Rare Human-Object Interactions Detection by Bridging Domain Gap","date":"2024-07-31","arxiv_id":"2407.21438","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-scene-understanding-through-object","slug":"dynamic-scene-understanding-through-object","title":"Dynamic Scene Understanding through Object-Centric Voxelization and Neural Rendering","date":"2024-07-30","arxiv_id":"2407.20908","repositories_listed":1,"syntology":null},{"url":"/paper/from-feature-importance-to-natural-language","slug":"from-feature-importance-to-natural-language","title":"From Feature Importance to Natural Language Explanations Using LLMs with RAG","date":"2024-07-30","arxiv_id":"2407.20990","repositories_listed":1,"syntology":null},{"url":"/paper/asi-seg-audio-driven-surgical-instrument","slug":"asi-seg-audio-driven-surgical-instrument","title":"ASI-Seg: Audio-Driven Surgical Instrument Segmentation with Surgeon Intention Understanding","date":"2024-07-28","arxiv_id":"2407.19435","repositories_listed":1,"syntology":null},{"url":"/paper/a-new-lightweight-hybrid-graph-convolutional","slug":"a-new-lightweight-hybrid-graph-convolutional","title":"A New Lightweight Hybrid Graph Convolutional Neural Network -- CNN Scheme for Scene Classification using Object Detection Inference","date":"2024-07-19","arxiv_id":"2407.14658","repositories_listed":1,"syntology":null},{"url":"/paper/mc-panda-mask-confidence-for-panoptic-domain","slug":"mc-panda-mask-confidence-for-panoptic-domain","title":"MC-PanDA: Mask Confidence for Panoptic Domain Adaptation","date":"2024-07-19","arxiv_id":"2407.14110","repositories_listed":1,"syntology":null},{"url":"/paper/general-geometry-aware-weakly-supervised-3d","slug":"general-geometry-aware-weakly-supervised-3d","title":"General Geometry-aware Weakly Supervised 3D Object Detection","date":"2024-07-18","arxiv_id":"2407.13748","repositories_listed":1,"syntology":null},{"url":"/paper/dual-hybrid-attention-network-for-specular","slug":"dual-hybrid-attention-network-for-specular","title":"Dual-Hybrid Attention Network for Specular Highlight Removal","date":"2024-07-17","arxiv_id":"2407.12255","repositories_listed":1,"syntology":null},{"url":"/paper/infonorm-mutual-information-shaping-of","slug":"infonorm-mutual-information-shaping-of","title":"InfoNorm: Mutual Information Shaping of Normals for Sparse-View Reconstruction","date":"2024-07-17","arxiv_id":"2407.12661","repositories_listed":1,"syntology":null},{"url":"/paper/no-train-all-gain-self-supervised-gradients","slug":"no-train-all-gain-self-supervised-gradients","title":"No Train, all Gain: Self-Supervised Gradients Improve Deep Frozen Representations","date":"2024-07-15","arxiv_id":"2407.10964","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/no-train-all-gain-self-supervised-gradients#ran","syntology_url":"https://syntology.ai/paper/2407.10964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10964"}},"official":{"repos":["waltersimoncini/fungivision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/shape2scene-3d-scene-representation-learning","slug":"shape2scene-3d-scene-representation-learning","title":"Shape2Scene: 3D Scene Representation Learning Through Pre-training on Shape Data","date":"2024-07-14","arxiv_id":"2407.10200","repositories_listed":1,"syntology":null},{"url":"/paper/swiss-dino-efficient-and-versatile-vision","slug":"swiss-dino-efficient-and-versatile-vision","title":"Swiss DINO: Efficient and Versatile Vision Framework for On-device Personal Object Search","date":"2024-07-10","arxiv_id":"2407.07541","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-framework-for-3d-scene","slug":"a-unified-framework-for-3d-scene","title":"A Unified Framework for 3D Scene Understanding","date":"2024-07-03","arxiv_id":"2407.03263","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-3d-scene#ran","syntology_url":"https://syntology.ai/paper/2407.03263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03263"}},"official":{"repos":["dk-liang/uniseg3d"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mtmamba-enhancing-multi-task-dense-scene","slug":"mtmamba-enhancing-multi-task-dense-scene","title":"MTMamba: Enhancing Multi-Task Dense Scene Understanding by Mamba-Based Decoders","date":"2024-07-02","arxiv_id":"2407.02228","repositories_listed":1,"syntology":null},{"url":"/paper/csfnet-a-cosine-similarity-fusion-network-for","slug":"csfnet-a-cosine-similarity-fusion-network-for","title":"CSFNet: A Cosine Similarity Fusion Network for Real-Time RGB-X Semantic Segmentation of Driving Scenes","date":"2024-07-01","arxiv_id":"2407.01328","repositories_listed":1,"syntology":null},{"url":"/paper/uni-dvps-unified-model-for-depth-aware-video","slug":"uni-dvps-unified-model-for-depth-aware-video","title":"Uni-DVPS: Unified Model for Depth-Aware Video Panoptic Segmentation","date":"2024-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/audiobench-a-universal-benchmark-for-audio","slug":"audiobench-a-universal-benchmark-for-audio","title":"AudioBench: A Universal Benchmark for Audio Large Language Models","date":"2024-06-23","arxiv_id":"2406.16020","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiobench-a-universal-benchmark-for-audio#ran","syntology_url":"https://syntology.ai/paper/2406.16020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16020"}},"official":{"repos":["audiollms/audiobench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stablesemantics-a-synthetic-language-vision","slug":"stablesemantics-a-synthetic-language-vision","title":"StableSemantics: A Synthetic Language-Vision Dataset of Semantic Representations in Naturalistic Images","date":"2024-06-19","arxiv_id":"2406.13735","repositories_listed":1,"syntology":null},{"url":"/paper/a-two-stage-masked-autoencoder-based-network","slug":"a-two-stage-masked-autoencoder-based-network","title":"A Two-Stage Masked Autoencoder Based Network for Indoor Depth Completion","date":"2024-06-14","arxiv_id":"2406.09792","repositories_listed":1,"syntology":null},{"url":"/paper/muirbench-a-comprehensive-benchmark-for","slug":"muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","arxiv_id":"2406.09411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/muirbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.09411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09411"}},"official":{"repos":["muirbench/MuirBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/category-level-neural-field-for","slug":"category-level-neural-field-for","title":"Category-level Neural Field for Reconstruction of Partially Observed Objects in Indoor Environment","date":"2024-06-12","arxiv_id":"2406.08176","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coped-advancing-multi-robot-collaborative","slug":"coped-advancing-multi-robot-collaborative","title":"CoPeD-Advancing Multi-Robot Collaborative Perception: A Comprehensive Dataset in Real-World Environments","date":"2024-05-23","arxiv_id":"2405.14731","repositories_listed":1,"syntology":null},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-appearances-material-segmentation-with","slug":"beyond-appearances-material-segmentation-with","title":"Beyond Appearances: Material Segmentation with Embedded Spectral Information from RGB-D imagery","date":"2024-05-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/grounded-3d-llm-with-referent-tokens","slug":"grounded-3d-llm-with-referent-tokens","title":"Grounded 3D-LLM with Referent Tokens","date":"2024-05-16","arxiv_id":"2405.10370","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-3d-llm-with-referent-tokens#ran","syntology_url":"https://syntology.ai/paper/2405.10370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10370"}},"official":{"repos":["OpenRobotLab/Grounded_3D-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/when-llms-step-into-the-3d-world-a-survey-and","slug":"when-llms-step-into-the-3d-world-a-survey-and","title":"When LLMs step into the 3D World: A Survey and Meta-Analysis of 3D Tasks via Multi-modal Large Language Models","date":"2024-05-16","arxiv_id":"2405.10255","repositories_listed":1,"syntology":null},{"url":"/paper/dtclmapper-dual-temporal-consistent-learning","slug":"dtclmapper-dual-temporal-consistent-learning","title":"DTCLMapper: Dual Temporal Consistent Learning for Vectorized HD Map Construction","date":"2024-05-09","arxiv_id":"2405.05518","repositories_listed":1,"syntology":null},{"url":"/paper/pre-trained-text-to-image-diffusion-models","slug":"pre-trained-text-to-image-diffusion-models","title":"Pre-trained Text-to-Image Diffusion Models Are Versatile Representation Learners for Control","date":"2024-05-09","arxiv_id":"2405.05852","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pre-trained-text-to-image-diffusion-models#ran","syntology_url":"https://syntology.ai/paper/2405.05852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05852"}},"official":{"repos":["ykarmesh/stable-control-representations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-data-efficient-3d-scene","slug":"multi-modal-data-efficient-3d-scene","title":"Multi-Modal Data-Efficient 3D Scene Understanding for Autonomous Driving","date":"2024-05-08","arxiv_id":"2405.05258","repositories_listed":1,"syntology":null},{"url":"/paper/openess-event-based-semantic-scene","slug":"openess-event-based-semantic-scene","title":"OpenESS: Event-based Semantic Scene Understanding with Open Vocabularies","date":"2024-05-08","arxiv_id":"2405.05259","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openess-event-based-semantic-scene#ran","syntology_url":"https://syntology.ai/paper/2405.05259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05259"}},"official":{"repos":["ldkong1205/openess"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bacs-background-aware-continual-semantic","slug":"bacs-background-aware-continual-semantic","title":"BACS: Background Aware Continual Semantic Segmentation","date":"2024-04-19","arxiv_id":"2404.13148","repositories_listed":1,"syntology":null},{"url":"/paper/spidepth-strengthened-pose-information-for","slug":"spidepth-strengthened-pose-information-for","title":"SPIdepth: Strengthened Pose Information for Self-supervised Monocular Depth Estimation","date":"2024-04-18","arxiv_id":"2404.12501","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":9,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 1 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spidepth-strengthened-pose-information-for#ran","syntology_url":"https://syntology.ai/paper/2404.12501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12501"}},"official":{"repos":["Lavreniuk/SPIdepth"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eclair-a-high-fidelity-aerial-lidar-dataset","slug":"eclair-a-high-fidelity-aerial-lidar-dataset","title":"ECLAIR: A High-Fidelity Aerial LiDAR Dataset for Semantic Segmentation","date":"2024-04-16","arxiv_id":"2404.10699","repositories_listed":1,"syntology":null},{"url":"/paper/pytorchgeonodes-enabling-differentiable-shape","slug":"pytorchgeonodes-enabling-differentiable-shape","title":"PyTorchGeoNodes: Enabling Differentiable Shape Programs for 3D Shape Reconstruction","date":"2024-04-16","arxiv_id":"2404.10620","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-object-dependencies-improving","slug":"mitigating-object-dependencies-improving","title":"Mitigating Object Dependencies: Improving Point Cloud Self-Supervised Learning through Object Exchange","date":"2024-04-11","arxiv_id":"2404.07504","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-object-dependencies-improving#ran","syntology_url":"https://syntology.ai/paper/2404.07504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07504"}},"official":{"repos":["yanhaowu/oessl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sigma-siamese-mamba-network-for-multi-modal","slug":"sigma-siamese-mamba-network-for-multi-modal","title":"Sigma: Siamese Mamba Network for Multi-Modal Semantic Segmentation","date":"2024-04-05","arxiv_id":"2404.04256","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sigma-siamese-mamba-network-for-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2404.04256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04256"}},"official":{"repos":["zifuwan/sigma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gov-nesf-generalizable-open-vocabulary-neural","slug":"gov-nesf-generalizable-open-vocabulary-neural","title":"GOV-NeSF: Generalizable Open-Vocabulary Neural Semantic Fields","date":"2024-04-01","arxiv_id":"2404.00931","repositories_listed":1,"syntology":null},{"url":"/paper/improving-visual-recognition-with","slug":"improving-visual-recognition-with","title":"Improving Visual Recognition with Hyperbolical Visual Hierarchy Mapping","date":"2024-04-01","arxiv_id":"2404.00974","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-visual-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2404.00974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00974"}},"official":{"repos":["kwonjunn01/hi-mapper"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nerf-mae-masked-autoencoders-for-self","slug":"nerf-mae-masked-autoencoders-for-self","title":"NeRF-MAE: Masked AutoEncoders for Self-Supervised 3D Representation Learning for Neural Radiance Fields","date":"2024-04-01","arxiv_id":"2404.01300","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nerf-mae-masked-autoencoders-for-self#ran","syntology_url":"https://syntology.ai/paper/2404.01300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01300"}},"official":{"repos":["zubair-irshad/NeRF-MAE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vsrd-instance-aware-volumetric-silhouette-1","slug":"vsrd-instance-aware-volumetric-silhouette-1","title":"VSRD: Instance-Aware Volumetric Silhouette Rendering for Weakly Supervised 3D Object Detection","date":"2024-03-29","arxiv_id":"2404.00149","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/vsrd-instance-aware-volumetric-silhouette-1#ran","syntology_url":"https://syntology.ai/paper/2404.00149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00149"}},"official":{"repos":["skmhrk1209/VSRD"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/object-pose-estimation-via-the-aggregation-of","slug":"object-pose-estimation-via-the-aggregation-of","title":"Object Pose Estimation via the Aggregation of Diffusion Features","date":"2024-03-27","arxiv_id":"2403.18791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-pose-estimation-via-the-aggregation-of#ran","syntology_url":"https://syntology.ai/paper/2403.18791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18791"}},"official":{"repos":["tianfu18/diff-feats-pose"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/calib3d-calibrating-model-preferences-for","slug":"calib3d-calibrating-model-preferences-for","title":"Calib3D: Calibrating Model Preferences for Reliable 3D Scene Understanding","date":"2024-03-25","arxiv_id":"2403.17010","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/calib3d-calibrating-model-preferences-for#ran","syntology_url":"https://syntology.ai/paper/2403.17010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17010"}},"official":{"repos":["ldkong1205/calib3d"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}}],"record_sha256":"d15360e69bfc6a1ca5e69ce671eb3b5ec7aaef7224290eb41da7440d1adbd3fb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}