{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/scene-understanding/papers/ran/1","list_of":"/task/scene-understanding","task":"Scene Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":208,"counts":{"archive_papers_tagged":1723,"with_a_code_link":720,"where_syntology_ran_a_sample":208,"not_listed_spam_title":0,"listed":1723,"listed_where_code_ran":208,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":182,"every_run_a_failure_of_syntologys_instrument":26,"listed_with_a_run_with_no_instrument_failure":182,"listed_every_run_a_failure_of_syntologys_instrument":26,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/scene-understanding/papers/ran/1","prev":null,"next":"/task/scene-understanding/papers/ran/2","papers":[{"url":"/paper/siu3r-simultaneous-scene-understanding-and-3d","slug":"siu3r-simultaneous-scene-understanding-and-3d","title":"SIU3R: Simultaneous Scene Understanding and 3D Reconstruction Beyond Feature Alignment","date":"2025-07-03","arxiv_id":"2507.02705","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/siu3r-simultaneous-scene-understanding-and-3d#ran","syntology_url":"https://syntology.ai/paper/2507.02705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02705"}},"official":{"repos":["WU-CVGL/SIU3R"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dip-unsupervised-dense-in-context-post","slug":"dip-unsupervised-dense-in-context-post","title":"DIP: Unsupervised Dense In-Context Post-training of Visual Representations","date":"2025-06-23","arxiv_id":"2506.18463","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dip-unsupervised-dense-in-context-post#ran","syntology_url":"https://syntology.ai/paper/2506.18463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18463"}},"official":{"repos":["sirkosophia/dip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-videos-for-3d-world-enhancing","slug":"learning-from-videos-for-3d-world-enhancing","title":"Learning from Videos for 3D World: Enhancing MLLMs with 3D Vision Geometry Priors","date":"2025-05-30","arxiv_id":"2505.24625","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/learning-from-videos-for-3d-world-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24625"}},"official":null}},{"url":"/paper/tackling-view-dependent-semantics-in-3d","slug":"tackling-view-dependent-semantics-in-3d","title":"Tackling View-Dependent Semantics in 3D Language Gaussian Splatting","date":"2025-05-30","arxiv_id":"2505.24746","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tackling-view-dependent-semantics-in-3d#ran","syntology_url":"https://syntology.ai/paper/2505.24746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24746"}},"official":{"repos":["sjtu-deepvisionlab/laga"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-empowered-embodied-agent-for-memory","slug":"llm-empowered-embodied-agent-for-memory","title":"LLM-Empowered Embodied Agent for Memory-Augmented Task Planning in Household Robotics","date":"2025-04-30","arxiv_id":"2504.21716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-empowered-embodied-agent-for-memory#ran","syntology_url":"https://syntology.ai/paper/2504.21716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21716"}},"official":{"repos":["marc1198/chat-hsr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scene-centric-unsupervised-panoptic","slug":"scene-centric-unsupervised-panoptic","title":"Scene-Centric Unsupervised Panoptic Segmentation","date":"2025-04-02","arxiv_id":"2504.01955","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/scene-centric-unsupervised-panoptic#ran","syntology_url":"https://syntology.ai/paper/2504.01955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01955"}},"official":{"repos":["visinf/cups"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","slug":"opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/opendrivevla-towards-end-to-end-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.23463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23463"}},"official":{"repos":["DriveVLA/OpenDriveVLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generating-realistic-3d-semantic","slug":"towards-generating-realistic-3d-semantic","title":"Towards Generating Realistic 3D Semantic Training Data for Autonomous Driving","date":"2025-03-27","arxiv_id":"2503.21449","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generating-realistic-3d-semantic#ran","syntology_url":"https://syntology.ai/paper/2503.21449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21449"}},"official":{"repos":["prbonn/3diss"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-and-uncertainty-aware","slug":"cross-modal-and-uncertainty-aware","title":"Cross-Modal and Uncertainty-Aware Agglomeration for Open-Vocabulary 3D Scene Understanding","date":"2025-03-20","arxiv_id":"2503.16707","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-and-uncertainty-aware#ran","syntology_url":"https://syntology.ai/paper/2503.16707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16707"}},"official":{"repos":["tyroneli/cua_o3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/psa-ssl-pose-and-size-aware-self-supervised","slug":"psa-ssl-pose-and-size-aware-self-supervised","title":"PSA-SSL: Pose and Size-aware Self-Supervised Learning on LiDAR Point Clouds","date":"2025-03-18","arxiv_id":"2503.13914","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/psa-ssl-pose-and-size-aware-self-supervised#ran","syntology_url":"https://syntology.ai/paper/2503.13914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13914"}},"official":null}},{"url":"/paper/a-data-centric-revisit-of-pre-trained-vision","slug":"a-data-centric-revisit-of-pre-trained-vision","title":"A Data-Centric Revisit of Pre-Trained Vision Models for Robot Learning","date":"2025-03-10","arxiv_id":"2503.06960","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-data-centric-revisit-of-pre-trained-vision#ran","syntology_url":"https://syntology.ai/paper/2503.06960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06960"}},"official":{"repos":["cvmi-lab/slotmim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hermes-a-unified-self-driving-world-model-for","slug":"hermes-a-unified-self-driving-world-model-for","title":"HERMES: A Unified Self-Driving World Model for Simultaneous 3D Scene Understanding and Generation","date":"2025-01-24","arxiv_id":"2501.14729","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hermes-a-unified-self-driving-world-model-for#ran","syntology_url":"https://syntology.ai/paper/2501.14729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14729"}},"official":{"repos":["lmd0311/hermes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/endochat-grounded-multimodal-large-language","slug":"endochat-grounded-multimodal-large-language","title":"EndoChat: Grounded Multimodal Large Language Model for Endoscopic Surgery","date":"2025-01-20","arxiv_id":"2501.11347","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/endochat-grounded-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2501.11347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11347"}},"official":{"repos":["gkw0010/endochat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt4scene-understand-3d-scenes-from-videos","slug":"gpt4scene-understand-3d-scenes-from-videos","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","date":"2025-01-02","arxiv_id":"2501.01428","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt4scene-understand-3d-scenes-from-videos#ran","syntology_url":"https://syntology.ai/paper/2501.01428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01428"}},"official":{"repos":["Qi-Zhangyang/GPT4Scene"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/3dgraphllm-combining-semantic-graphs-and","slug":"3dgraphllm-combining-semantic-graphs-and","title":"3DGraphLLM: Combining Semantic Graphs and Large Language Models for 3D Scene Understanding","date":"2024-12-24","arxiv_id":"2412.18450","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3dgraphllm-combining-semantic-graphs-and#ran","syntology_url":"https://syntology.ai/paper/2412.18450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18450"}},"official":{"repos":["cognitiveaisystems/3dgraphllm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dense-audio-visual-event-localization-under","slug":"dense-audio-visual-event-localization-under","title":"Dense Audio-Visual Event Localization under Cross-Modal Consistency and Multi-Temporal Granularity Collaboration","date":"2024-12-17","arxiv_id":"2412.12628","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dense-audio-visual-event-localization-under#ran","syntology_url":"https://syntology.ai/paper/2412.12628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12628"}},"official":{"repos":["zzhhfut/ccnet-aaai2025"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-dino-foresight-looking-into-the-future","slug":"texttt-dino-foresight-looking-into-the-future","title":"$\\texttt{DINO-Foresight}$: Looking into the Future with DINO","date":"2024-12-16","arxiv_id":"2412.11673","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/texttt-dino-foresight-looking-into-the-future#ran","syntology_url":"https://syntology.ai/paper/2412.11673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11673"}},"official":{"repos":["sta8is/dino-foresight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wisead-knowledge-augmented-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.09951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09951"}},"official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stag-1-towards-realistic-4d-driving","slug":"stag-1-towards-realistic-4d-driving","title":"Stag-1: Towards Realistic 4D Driving Simulation with Video Generation Model","date":"2024-12-06","arxiv_id":"2412.05280","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stag-1-towards-realistic-4d-driving#ran","syntology_url":"https://syntology.ai/paper/2412.05280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05280"}},"official":{"repos":["wzzheng/stag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/embodiedocc-embodied-3d-occupancy-prediction","slug":"embodiedocc-embodied-3d-occupancy-prediction","title":"EmbodiedOcc: Embodied 3D Occupancy Prediction for Vision-based Online Scene Understanding","date":"2024-12-05","arxiv_id":"2412.04380","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodiedocc-embodied-3d-occupancy-prediction#ran","syntology_url":"https://syntology.ai/paper/2412.04380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04380"}},"official":{"repos":["ykiwu/embodiedocc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lscenellm-enhancing-large-3d-scene","slug":"lscenellm-enhancing-large-3d-scene","title":"LSceneLLM: Enhancing Large 3D Scene Understanding Using Adaptive Visual Preferences","date":"2024-12-02","arxiv_id":"2412.01292","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":4,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lscenellm-enhancing-large-3d-scene#ran","syntology_url":"https://syntology.ai/paper/2412.01292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01292"}},"official":{"repos":["Hoyyyaard/LSceneLLM"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bootstraping-clustering-of-gaussians-for-view","slug":"bootstraping-clustering-of-gaussians-for-view","title":"Bootstraping Clustering of Gaussians for View-consistent 3D Scene Understanding","date":"2024-11-29","arxiv_id":"2411.19551","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bootstraping-clustering-of-gaussians-for-view#ran","syntology_url":"https://syntology.ai/paper/2411.19551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19551"}},"official":{"repos":["wb014/FreeGS"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geobench-vlm-benchmarking-vision-language","slug":"geobench-vlm-benchmarking-vision-language","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","date":"2024-11-28","arxiv_id":"2411.19325","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/geobench-vlm-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.19325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19325"}},"official":{"repos":["the-ai-alliance/geo-bench-vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gaussianpretrain-a-simple-unified-3d-gaussian","slug":"gaussianpretrain-a-simple-unified-3d-gaussian","title":"GaussianPretrain: A Simple Unified 3D Gaussian Representation for Visual Pre-training in Autonomous Driving","date":"2024-11-19","arxiv_id":"2411.12452","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gaussianpretrain-a-simple-unified-3d-gaussian#ran","syntology_url":"https://syntology.ai/paper/2411.12452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12452"}},"official":{"repos":["public-bots/gaussianpretrain"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/senna-bridging-large-vision-language-models","slug":"senna-bridging-large-vision-language-models","title":"Senna: Bridging Large Vision-Language Models and End-to-End Autonomous Driving","date":"2024-10-29","arxiv_id":"2410.22313","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/senna-bridging-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.22313","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22313"}},"official":{"repos":["hustvl/senna"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d","slug":"vlm-grounder-a-vlm-agent-for-zero-shot-3d","title":"VLM-Grounder: A VLM Agent for Zero-Shot 3D Visual Grounding","date":"2024-10-17","arxiv_id":"2410.13860","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vlm-grounder-a-vlm-agent-for-zero-shot-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13860"}},"official":{"repos":["openrobotlab/vlm-grounder"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/arkit-labelmaker-a-new-scale-for-indoor-3d","slug":"arkit-labelmaker-a-new-scale-for-indoor-3d","title":"ARKit LabelMaker: A New Scale for Indoor 3D Scene Understanding","date":"2024-10-17","arxiv_id":"2410.13924","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/arkit-labelmaker-a-new-scale-for-indoor-3d#ran","syntology_url":"https://syntology.ai/paper/2410.13924","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13924"}},"official":{"repos":["cvg/labelmaker"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/procedure-aware-surgical-video-language","slug":"procedure-aware-surgical-video-language","title":"Procedure-Aware Surgical Video-language Pretraining with Hierarchical Knowledge Augmentation","date":"2024-09-30","arxiv_id":"2410.00263","repositories_listed":2,"syntology":{"n":15,"n_ran":9,"n_constructed":1,"n_ran_checked":8,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":15,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/procedure-aware-surgical-video-language#ran","syntology_url":"https://syntology.ai/paper/2410.00263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00263"}},"official":null}},{"url":"/paper/hi-slam-scaling-up-semantics-in-slam-with-a","slug":"hi-slam-scaling-up-semantics-in-slam-with-a","title":"Hier-SLAM: Scaling-up Semantics in SLAM with a Hierarchically Categorical Gaussian Splatting","date":"2024-09-19","arxiv_id":"2409.12518","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hi-slam-scaling-up-semantics-in-slam-with-a#ran","syntology_url":"https://syntology.ai/paper/2409.12518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12518"}},"official":{"repos":["LeeBY68/Hier-SLAM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/primedepth-efficient-monocular-depth","slug":"primedepth-efficient-monocular-depth","title":"PrimeDepth: Efficient Monocular Depth Estimation with a Stable Diffusion Preimage","date":"2024-09-13","arxiv_id":"2409.09144","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":3,"n_no_contract":5,"n_pointer_only":6,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 3 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/primedepth-efficient-monocular-depth#ran","syntology_url":"https://syntology.ai/paper/2409.09144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.09144"}},"official":{"repos":["vislearn/PrimeDepth"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/loss-distillation-via-gradient-matching-for","slug":"loss-distillation-via-gradient-matching-for","title":"Loss Distillation via Gradient Matching for Point Cloud Completion with Weighted Chamfer Distance","date":"2024-09-10","arxiv_id":"2409.06171","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/loss-distillation-via-gradient-matching-for#ran","syntology_url":"https://syntology.ai/paper/2409.06171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.06171"}},"official":{"repos":["zhang-vislab/iros2024-lossdistillationweightedcd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lexicon3d-probing-visual-foundation-models","slug":"lexicon3d-probing-visual-foundation-models","title":"Lexicon3D: Probing Visual Foundation Models for Complex 3D Scene Understanding","date":"2024-09-05","arxiv_id":"2409.03757","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lexicon3d-probing-visual-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2409.03757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03757"}},"official":null}},{"url":"/paper/adaptvision-dynamic-input-scaling-in-mllms","slug":"adaptvision-dynamic-input-scaling-in-mllms","title":"AdaptVision: Dynamic Input Scaling in MLLMs for Versatile Scene Understanding","date":"2024-08-30","arxiv_id":"2408.16986","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptvision-dynamic-input-scaling-in-mllms#ran","syntology_url":"https://syntology.ai/paper/2408.16986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16986"}},"official":{"repos":["harrytea/adaptvision"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/urbench-a-comprehensive-benchmark-for","slug":"urbench-a-comprehensive-benchmark-for","title":"UrBench: A Comprehensive Benchmark for Evaluating Large Multimodal Models in Multi-View Urban Scenarios","date":"2024-08-30","arxiv_id":"2408.17267","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/urbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2408.17267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.17267"}},"official":{"repos":["opendatalab/urbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rsteller-scaling-up-visual-language-modeling","slug":"rsteller-scaling-up-visual-language-modeling","title":"RSTeller: Scaling Up Visual Language Modeling in Remote Sensing with Rich Linguistic Semantics from Openly Available Data and Large Language Models","date":"2024-08-27","arxiv_id":"2408.14744","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rsteller-scaling-up-visual-language-modeling#ran","syntology_url":"https://syntology.ai/paper/2408.14744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14744"}},"official":{"repos":["slytheringe/rsteller"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mtmamba-enhancing-multi-task-dense-scene-1","slug":"mtmamba-enhancing-multi-task-dense-scene-1","title":"MTMamba++: Enhancing Multi-Task Dense Scene Understanding via Mamba-Based Decoders","date":"2024-08-27","arxiv_id":"2408.15101","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtmamba-enhancing-multi-task-dense-scene-1#ran","syntology_url":"https://syntology.ai/paper/2408.15101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15101"}},"official":{"repos":["envision-research/mtmamba"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openscan-a-benchmark-for-generalized-open","slug":"openscan-a-benchmark-for-generalized-open","title":"OpenScan: A Benchmark for Generalized Open-Vocabulary 3D Scene Understanding","date":"2024-08-20","arxiv_id":"2408.11030","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openscan-a-benchmark-for-generalized-open#ran","syntology_url":"https://syntology.ai/paper/2408.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11030"}},"official":{"repos":["youjunzhao/openscan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deepinteraction-multi-modality-interaction","slug":"deepinteraction-multi-modality-interaction","title":"DeepInteraction++: Multi-Modality Interaction for Autonomous Driving","date":"2024-08-09","arxiv_id":"2408.05075","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepinteraction-multi-modality-interaction#ran","syntology_url":"https://syntology.ai/paper/2408.05075","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.05075"}},"official":{"repos":["fudan-zvg/deepinteraction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-3d-scene-understanding-via","slug":"open-vocabulary-3d-scene-understanding-via","title":"Open Vocabulary 3D Scene Understanding via Geometry Guided Self-Distillation","date":"2024-07-18","arxiv_id":"2407.13362","repositories_listed":0,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/open-vocabulary-3d-scene-understanding-via#ran","syntology_url":"https://syntology.ai/paper/2407.13362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13362"}},"official":null}},{"url":"/paper/no-train-all-gain-self-supervised-gradients","slug":"no-train-all-gain-self-supervised-gradients","title":"No Train, all Gain: Self-Supervised Gradients Improve Deep Frozen Representations","date":"2024-07-15","arxiv_id":"2407.10964","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/no-train-all-gain-self-supervised-gradients#ran","syntology_url":"https://syntology.ai/paper/2407.10964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10964"}},"official":{"repos":["waltersimoncini/fungivision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pareto-low-rank-adapters-efficient-multi-task","slug":"pareto-low-rank-adapters-efficient-multi-task","title":"Pareto Low-Rank Adapters: Efficient Multi-Task Learning with Preferences","date":"2024-07-10","arxiv_id":"2407.08056","repositories_listed":0,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/pareto-low-rank-adapters-efficient-multi-task#ran","syntology_url":"https://syntology.ai/paper/2407.08056","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08056"}},"official":null}},{"url":"/paper/a-unified-framework-for-3d-scene","slug":"a-unified-framework-for-3d-scene","title":"A Unified Framework for 3D Scene Understanding","date":"2024-07-03","arxiv_id":"2407.03263","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-framework-for-3d-scene#ran","syntology_url":"https://syntology.ai/paper/2407.03263","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03263"}},"official":{"repos":["dk-liang/uniseg3d"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audiobench-a-universal-benchmark-for-audio","slug":"audiobench-a-universal-benchmark-for-audio","title":"AudioBench: A Universal Benchmark for Audio Large Language Models","date":"2024-06-23","arxiv_id":"2406.16020","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiobench-a-universal-benchmark-for-audio#ran","syntology_url":"https://syntology.ai/paper/2406.16020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16020"}},"official":{"repos":["audiollms/audiobench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/muirbench-a-comprehensive-benchmark-for","slug":"muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","arxiv_id":"2406.09411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/muirbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.09411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09411"}},"official":{"repos":["muirbench/MuirBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/4d-panoptic-scene-graph-generation-1","slug":"4d-panoptic-scene-graph-generation-1","title":"4D Panoptic Scene Graph Generation","date":"2024-05-16","arxiv_id":"2405.10305","repositories_listed":3,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":10,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/4d-panoptic-scene-graph-generation-1#ran","syntology_url":"https://syntology.ai/paper/2405.10305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10305"}},"official":{"repos":["jingkang50/psg4d","Jingkang50/OpenPSG","jingkang50/openpvsg"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-3d-llm-with-referent-tokens","slug":"grounded-3d-llm-with-referent-tokens","title":"Grounded 3D-LLM with Referent Tokens","date":"2024-05-16","arxiv_id":"2405.10370","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-3d-llm-with-referent-tokens#ran","syntology_url":"https://syntology.ai/paper/2405.10370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10370"}},"official":{"repos":["OpenRobotLab/Grounded_3D-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-trained-text-to-image-diffusion-models","slug":"pre-trained-text-to-image-diffusion-models","title":"Pre-trained Text-to-Image Diffusion Models Are Versatile Representation Learners for Control","date":"2024-05-09","arxiv_id":"2405.05852","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pre-trained-text-to-image-diffusion-models#ran","syntology_url":"https://syntology.ai/paper/2405.05852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05852"}},"official":{"repos":["ykarmesh/stable-control-representations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/openess-event-based-semantic-scene","slug":"openess-event-based-semantic-scene","title":"OpenESS: Event-based Semantic Scene Understanding with Open Vocabularies","date":"2024-05-08","arxiv_id":"2405.05259","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openess-event-based-semantic-scene#ran","syntology_url":"https://syntology.ai/paper/2405.05259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05259"}},"official":{"repos":["ldkong1205/openess"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/spidepth-strengthened-pose-information-for","slug":"spidepth-strengthened-pose-information-for","title":"SPIdepth: Strengthened Pose Information for Self-supervised Monocular Depth Estimation","date":"2024-04-18","arxiv_id":"2404.12501","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":9,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 1 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/spidepth-strengthened-pose-information-for#ran","syntology_url":"https://syntology.ai/paper/2404.12501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12501"}},"official":{"repos":["Lavreniuk/SPIdepth"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-object-dependencies-improving","slug":"mitigating-object-dependencies-improving","title":"Mitigating Object Dependencies: Improving Point Cloud Self-Supervised Learning through Object Exchange","date":"2024-04-11","arxiv_id":"2404.07504","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-object-dependencies-improving#ran","syntology_url":"https://syntology.ai/paper/2404.07504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07504"}},"official":{"repos":["yanhaowu/oessl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sigma-siamese-mamba-network-for-multi-modal","slug":"sigma-siamese-mamba-network-for-multi-modal","title":"Sigma: Siamese Mamba Network for Multi-Modal Semantic Segmentation","date":"2024-04-05","arxiv_id":"2404.04256","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sigma-siamese-mamba-network-for-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2404.04256","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04256"}},"official":{"repos":["zifuwan/sigma"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-visual-recognition-with","slug":"improving-visual-recognition-with","title":"Improving Visual Recognition with Hyperbolical Visual Hierarchy Mapping","date":"2024-04-01","arxiv_id":"2404.00974","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-visual-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2404.00974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00974"}},"official":{"repos":["kwonjunn01/hi-mapper"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/nerf-mae-masked-autoencoders-for-self","slug":"nerf-mae-masked-autoencoders-for-self","title":"NeRF-MAE: Masked AutoEncoders for Self-Supervised 3D Representation Learning for Neural Radiance Fields","date":"2024-04-01","arxiv_id":"2404.01300","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nerf-mae-masked-autoencoders-for-self#ran","syntology_url":"https://syntology.ai/paper/2404.01300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01300"}},"official":{"repos":["zubair-irshad/NeRF-MAE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/object-pose-estimation-via-the-aggregation-of","slug":"object-pose-estimation-via-the-aggregation-of","title":"Object Pose Estimation via the Aggregation of Diffusion Features","date":"2024-03-27","arxiv_id":"2403.18791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-pose-estimation-via-the-aggregation-of#ran","syntology_url":"https://syntology.ai/paper/2403.18791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18791"}},"official":{"repos":["tianfu18/diff-feats-pose"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimizing-lidar-placements-for-robust","slug":"optimizing-lidar-placements-for-robust","title":"Is Your LiDAR Placement Optimized for 3D Scene Understanding?","date":"2024-03-25","arxiv_id":"2403.17009","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimizing-lidar-placements-for-robust#ran","syntology_url":"https://syntology.ai/paper/2403.17009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17009"}},"official":{"repos":["ywyeli/place3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omni-recon-towards-general-purpose-neural","slug":"omni-recon-towards-general-purpose-neural","title":"Omni-Recon: Harnessing Image-based Rendering for General-Purpose Neural Radiance Fields","date":"2024-03-17","arxiv_id":"2403.11131","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/omni-recon-towards-general-purpose-neural#ran","syntology_url":"https://syntology.ai/paper/2403.11131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11131"}},"official":{"repos":["GATECH-EIC/Omni-Recon"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/moai-mixture-of-all-intelligence-for-large","slug":"moai-mixture-of-all-intelligence-for-large","title":"MoAI: Mixture of All Intelligence for Large Language and Vision Models","date":"2024-03-12","arxiv_id":"2403.07508","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moai-mixture-of-all-intelligence-for-large#ran","syntology_url":"https://syntology.ai/paper/2403.07508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07508"}},"official":{"repos":["ByungKwanLee/MoAI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-understanding-of-driving-scenarios","slug":"embodied-understanding-of-driving-scenarios","title":"Embodied Understanding of Driving Scenarios","date":"2024-03-07","arxiv_id":"2403.04593","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/embodied-understanding-of-driving-scenarios#ran","syntology_url":"https://syntology.ai/paper/2403.04593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04593"}},"official":{"repos":["opendrivelab/elm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/swin3d-effective-multi-source-pretraining-for","slug":"swin3d-effective-multi-source-pretraining-for","title":"Swin3D++: Effective Multi-Source Pretraining for 3D Indoor Scene Understanding","date":"2024-02-22","arxiv_id":"2402.14215","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/swin3d-effective-multi-source-pretraining-for#ran","syntology_url":"https://syntology.ai/paper/2402.14215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14215"}},"official":{"repos":["microsoft/swin3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","slug":"prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prismatic-vlms-investigating-the-design-space#ran","syntology_url":"https://syntology.ai/paper/2402.07865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07865"}},"official":{"repos":["tri-ml/prismatic-vlms","tri-ml/vlm-evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sgs-slam-semantic-gaussian-splatting-for","slug":"sgs-slam-semantic-gaussian-splatting-for","title":"SGS-SLAM: Semantic Gaussian Splatting For Neural Dense SLAM","date":"2024-02-05","arxiv_id":"2402.03246","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sgs-slam-semantic-gaussian-splatting-for#ran","syntology_url":"https://syntology.ai/paper/2402.03246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03246"}},"official":{"repos":["shuhongll/sgs-slam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unim-ov3d-uni-modality-open-vocabulary-3d","slug":"unim-ov3d-uni-modality-open-vocabulary-3d","title":"UniM-OV3D: Uni-Modality Open-Vocabulary 3D Scene Understanding with Fine-Grained Feature Representation","date":"2024-01-21","arxiv_id":"2401.11395","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/unim-ov3d-uni-modality-open-vocabulary-3d#ran","syntology_url":"https://syntology.ai/paper/2401.11395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11395"}},"official":{"repos":["hithqd/unim-ov3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pixel-wise-recognition-for-holistic-surgical","slug":"pixel-wise-recognition-for-holistic-surgical","title":"Pixel-Wise Recognition for Holistic Surgical Scene Understanding","date":"2024-01-20","arxiv_id":"2401.11174","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pixel-wise-recognition-for-holistic-surgical#ran","syntology_url":"https://syntology.ai/paper/2401.11174","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11174"}},"official":{"repos":["bcv-uniandes/grasp","bcv-uniandes/matis","bcv-uniandes/tapir"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/icgnet-a-unified-approach-for-instance","slug":"icgnet-a-unified-approach-for-instance","title":"ICGNet: A Unified Approach for Instance-Centric Grasping","date":"2024-01-18","arxiv_id":"2401.09939","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/icgnet-a-unified-approach-for-instance#ran","syntology_url":"https://syntology.ai/paper/2401.09939","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09939"}},"official":{"repos":["renezurbruegg/icg_benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/garfield-group-anything-with-radiance-fields","slug":"garfield-group-anything-with-radiance-fields","title":"GARField: Group Anything with Radiance Fields","date":"2024-01-17","arxiv_id":"2401.09419","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/garfield-group-anything-with-radiance-fields#ran","syntology_url":"https://syntology.ai/paper/2401.09419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09419"}},"official":{"repos":["chungmin99/garfield"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rsud20k-a-dataset-for-road-scene","slug":"rsud20k-a-dataset-for-road-scene","title":"RSUD20K: A Dataset for Road Scene Understanding In Autonomous Driving","date":"2024-01-14","arxiv_id":"2401.07322","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rsud20k-a-dataset-for-road-scene#ran","syntology_url":"https://syntology.ai/paper/2401.07322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07322"}},"official":{"repos":["hasibzunair/rsud20k"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/3dmit-3d-multi-modal-instruction-tuning-for","slug":"3dmit-3d-multi-modal-instruction-tuning-for","title":"3DMIT: 3D Multi-modal Instruction Tuning for Scene Understanding","date":"2024-01-06","arxiv_id":"2401.03201","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/3dmit-3d-multi-modal-instruction-tuning-for#ran","syntology_url":"https://syntology.ai/paper/2401.03201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03201"}},"official":{"repos":["staymylove/3DMIT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/open3dis-open-vocabulary-3d-instance","slug":"open3dis-open-vocabulary-3d-instance","title":"Open3DIS: Open-Vocabulary 3D Instance Segmentation with 2D Mask Guidance","date":"2023-12-17","arxiv_id":"2312.10671","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open3dis-open-vocabulary-3d-instance#ran","syntology_url":"https://syntology.ai/paper/2312.10671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10671"}},"official":{"repos":["VinAIResearch/Open3DIS"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/living-scenes-multi-object-relocalization-and","slug":"living-scenes-multi-object-relocalization-and","title":"Living Scenes: Multi-object Relocalization and Reconstruction in Changing 3D Environments","date":"2023-12-14","arxiv_id":"2312.09138","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/living-scenes-multi-object-relocalization-and#ran","syntology_url":"https://syntology.ai/paper/2312.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09138"}},"official":{"repos":["GradientSpaces/LivingScenes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chat-3d-v2-bridging-3d-scene-and-large","slug":"chat-3d-v2-bridging-3d-scene-and-large","title":"Chat-Scene: Bridging 3D Scene and Large Language Models with Object Identifiers","date":"2023-12-13","arxiv_id":"2312.08168","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chat-3d-v2-bridging-3d-scene-and-large#ran","syntology_url":"https://syntology.ai/paper/2312.08168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08168"}},"official":{"repos":["chat-3d/chat-3d-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/x4d-sceneformer-enhanced-scene-understanding","slug":"x4d-sceneformer-enhanced-scene-understanding","title":"X4D-SceneFormer: Enhanced Scene Understanding on 4D Point Cloud Videos through Cross-modal Knowledge Transfer","date":"2023-12-12","arxiv_id":"2312.07378","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x4d-sceneformer-enhanced-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2312.07378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07378"}},"official":{"repos":["jinglinglingling/x4d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-ss3d-diffusion-model-for-semi-1","slug":"diffusion-ss3d-diffusion-model-for-semi-1","title":"Diffusion-SS3D: Diffusion Model for Semi-supervised 3D Object Detection","date":"2023-12-05","arxiv_id":"2312.02966","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/diffusion-ss3d-diffusion-model-for-semi-1#ran","syntology_url":"https://syntology.ai/paper/2312.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02966"}},"official":{"repos":["luluho1208/diffusion-ss3d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/repurposing-diffusion-based-image-generators","slug":"repurposing-diffusion-based-image-generators","title":"Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation","date":"2023-12-04","arxiv_id":"2312.02145","repositories_listed":4,"syntology":{"n":26,"n_ran":22,"n_constructed":0,"n_ran_checked":19,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":3,"n_no_contract":16,"n_pointer_only":5,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 3 violated, 16 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/repurposing-diffusion-based-image-generators#ran","syntology_url":"https://syntology.ai/paper/2312.02145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02145"}},"official":{"repos":["prs-eth/marigold"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gaussian-grouping-segment-and-edit-anything","slug":"gaussian-grouping-segment-and-edit-anything","title":"Gaussian Grouping: Segment and Edit Anything in 3D Scenes","date":"2023-12-01","arxiv_id":"2312.00732","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gaussian-grouping-segment-and-edit-anything#ran","syntology_url":"https://syntology.ai/paper/2312.00732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00732"}},"official":{"repos":["lkeab/gaussian-grouping"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sampro3d-locating-sam-prompts-in-3d-for-zero","slug":"sampro3d-locating-sam-prompts-in-3d-for-zero","title":"SAMPro3D: Locating SAM Prompts in 3D for Zero-Shot Scene Segmentation","date":"2023-11-29","arxiv_id":"2311.17707","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/sampro3d-locating-sam-prompts-in-3d-for-zero#ran","syntology_url":"https://syntology.ai/paper/2311.17707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17707"}},"official":{"repos":["GAP-LAB-CUHK-SZ/SAMPro3D"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/panoptic-video-scene-graph-generation-1","slug":"panoptic-video-scene-graph-generation-1","title":"Panoptic Video Scene Graph Generation","date":"2023-11-28","arxiv_id":"2311.17058","repositories_listed":3,"syntology":{"n":11,"n_ran":11,"n_constructed":2,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/panoptic-video-scene-graph-generation-1#ran","syntology_url":"https://syntology.ai/paper/2311.17058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17058"}},"official":{"repos":["jingkang50/openpvsg","lilydaytoy/openpvsg","lilydaytoy/pvsgannotation"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":2,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tpsence-towards-artifact-free-realistic-rain","slug":"tpsence-towards-artifact-free-realistic-rain","title":"TPSeNCE: Towards Artifact-Free Realistic Rain Generation for Deraining and Object Detection in Rain","date":"2023-11-01","arxiv_id":"2311.00660","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tpsence-towards-artifact-free-realistic-rain#ran","syntology_url":"https://syntology.ai/paper/2311.00660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00660"}},"official":{"repos":["shenzheng2000/tpsence"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/talk2bev-language-enhanced-bird-s-eye-view","slug":"talk2bev-language-enhanced-bird-s-eye-view","title":"Talk2BEV: Language-enhanced Bird's-eye View Maps for Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.02251","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/talk2bev-language-enhanced-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2310.02251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02251"}},"official":null}},{"url":"/paper/adaptive-visual-scene-understanding","slug":"adaptive-visual-scene-understanding","title":"Adaptive Visual Scene Understanding: Incremental Scene Graph Generation","date":"2023-10-02","arxiv_id":"2310.01636","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-visual-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2310.01636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01636"}},"official":{"repos":["zhanglab-deepneurocoglab/csegg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/panopticndt-efficient-and-robust-panoptic","slug":"panopticndt-efficient-and-robust-panoptic","title":"PanopticNDT: Efficient and Robust Panoptic Mapping","date":"2023-09-24","arxiv_id":"2309.13635","repositories_listed":4,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/panopticndt-efficient-and-robust-panoptic#ran","syntology_url":"https://syntology.ai/paper/2309.13635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13635"}},"official":{"repos":["tui-nicr/panoptic-mapping"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/shape-anchor-guided-holistic-indoor-scene","slug":"shape-anchor-guided-holistic-indoor-scene","title":"Shape Anchor Guided Holistic Indoor Scene Understanding","date":"2023-09-20","arxiv_id":"2309.11133","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/shape-anchor-guided-holistic-indoor-scene#ran","syntology_url":"https://syntology.ai/paper/2309.11133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11133"}},"official":{"repos":["Geo-Tell/AncRec"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi3drefer-grounding-text-description-to","slug":"multi3drefer-grounding-text-description-to","title":"Multi3DRefer: Grounding Text Description to Multiple 3D Objects","date":"2023-09-11","arxiv_id":"2309.05251","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi3drefer-grounding-text-description-to#ran","syntology_url":"https://syntology.ai/paper/2309.05251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05251"}},"official":{"repos":["3dlg-hcvc/M3DRef-CLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/openins3d-snap-and-lookup-for-3d-open","slug":"openins3d-snap-and-lookup-for-3d-open","title":"OpenIns3D: Snap and Lookup for 3D Open-vocabulary Instance Segmentation","date":"2023-09-01","arxiv_id":"2309.00616","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openins3d-snap-and-lookup-for-3d-open#ran","syntology_url":"https://syntology.ai/paper/2309.00616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00616"}},"official":{"repos":["Pointcept/OpenIns3D"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/summit-source-free-adaptation-of-uni-modal","slug":"summit-source-free-adaptation-of-uni-modal","title":"SUMMIT: Source-Free Adaptation of Uni-Modal Models to Multi-Modal Targets","date":"2023-08-23","arxiv_id":"2308.11880","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/summit-source-free-adaptation-of-uni-modal#ran","syntology_url":"https://syntology.ai/paper/2308.11880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11880"}},"official":{"repos":["csimo005/summit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scannet-a-high-fidelity-dataset-of-3d-indoor","slug":"scannet-a-high-fidelity-dataset-of-3d-indoor","title":"ScanNet++: A High-Fidelity Dataset of 3D Indoor Scenes","date":"2023-08-22","arxiv_id":"2308.11417","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scannet-a-high-fidelity-dataset-of-3d-indoor#ran","syntology_url":"https://syntology.ai/paper/2308.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11417"}},"official":null}},{"url":"/paper/vision-relation-transformer-for-unbiased","slug":"vision-relation-transformer-for-unbiased","title":"Vision Relation Transformer for Unbiased Scene Graph Generation","date":"2023-08-18","arxiv_id":"2308.09472","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vision-relation-transformer-for-unbiased#ran","syntology_url":"https://syntology.ai/paper/2308.09472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09472"}},"official":{"repos":["visinf/veto"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taskexpert-dynamically-assembling-multi-task","slug":"taskexpert-dynamically-assembling-multi-task","title":"TaskExpert: Dynamically Assembling Multi-Task Representations with Memorial Mixture-of-Experts","date":"2023-07-28","arxiv_id":"2307.15324","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/taskexpert-dynamically-assembling-multi-task#ran","syntology_url":"https://syntology.ai/paper/2307.15324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15324"}},"official":{"repos":["prismformore/multi-task-transformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/human-centric-scene-understanding-for-3d-1","slug":"human-centric-scene-understanding-for-3d-1","title":"Human-centric Scene Understanding for 3D Large-scale Scenarios","date":"2023-07-26","arxiv_id":"2307.14392","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-scene-understanding-for-3d-1#ran","syntology_url":"https://syntology.ai/paper/2307.14392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14392"}},"official":{"repos":["4dvlab/hucenlife"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-distillation-for-continual","slug":"revisiting-distillation-for-continual","title":"Revisiting Distillation for Continual Learning on Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-22","arxiv_id":"2307.12045","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-distillation-for-continual#ran","syntology_url":"https://syntology.ai/paper/2307.12045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12045"}},"official":{"repos":["longbai1006/cs-vqla"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cpcm-contextual-point-cloud-modeling-for","slug":"cpcm-contextual-point-cloud-modeling-for","title":"CPCM: Contextual Point Cloud Modeling for Weakly-supervised Point Cloud Semantic Segmentation","date":"2023-07-19","arxiv_id":"2307.10316","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cpcm-contextual-point-cloud-modeling-for#ran","syntology_url":"https://syntology.ai/paper/2307.10316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10316"}},"official":{"repos":["lizhaoliu-Lec/CPCM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/physion-evaluating-physical-scene","slug":"physion-evaluating-physical-scene","title":"Physion++: Evaluating Physical Scene Understanding that Requires Online Inference of Different Physical Properties","date":"2023-06-27","arxiv_id":"2306.15668","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/physion-evaluating-physical-scene#ran","syntology_url":"https://syntology.ai/paper/2306.15668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15668"}},"official":null}},{"url":"/paper/openmask3d-open-vocabulary-3d-instance","slug":"openmask3d-open-vocabulary-3d-instance","title":"OpenMask3D: Open-Vocabulary 3D Instance Segmentation","date":"2023-06-23","arxiv_id":"2306.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openmask3d-open-vocabulary-3d-instance#ran","syntology_url":"https://syntology.ai/paper/2306.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13631"}},"official":{"repos":["OpenMask3D/openmask3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/estimating-generic-3d-room-structures-from-2d-1","slug":"estimating-generic-3d-room-structures-from-2d-1","title":"Estimating Generic 3D Room Structures from 2D Annotations","date":"2023-06-15","arxiv_id":"2306.09077","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/estimating-generic-3d-room-structures-from-2d-1#ran","syntology_url":"https://syntology.ai/paper/2306.09077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09077"}},"official":{"repos":["google-research/cad-estate"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/snap-self-supervised-neural-maps-for-visual-1","slug":"snap-self-supervised-neural-maps-for-visual-1","title":"SNAP: Self-Supervised Neural Maps for Visual Positioning and Semantic Understanding","date":"2023-06-08","arxiv_id":"2306.05407","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/snap-self-supervised-neural-maps-for-visual-1#ran","syntology_url":"https://syntology.ai/paper/2306.05407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05407"}},"official":{"repos":["google-research/snap"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"283d47259217b1b2c873b39fa4e6ce4c2f5e529f578b977b2020cb714514f3ff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}