{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-alignment/papers/ran/1","list_of":"/task/cross-modal-alignment","task":"cross-modal alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,47],"of":47,"counts":{"archive_papers_tagged":342,"with_a_code_link":151,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":342,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":41,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":41,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-alignment/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/skywork-r1v3-technical-report","slug":"skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","arxiv_id":"2507.06167","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skywork-r1v3-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.06167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06167"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/desta2-5-audio-toward-general-purpose-large","slug":"desta2-5-audio-toward-general-purpose-large","title":"DeSTA2.5-Audio: Toward General-Purpose Large Audio Language Model with Self-Generated Cross-Modal Alignment","date":"2025-07-03","arxiv_id":"2507.02768","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/desta2-5-audio-toward-general-purpose-large#ran","syntology_url":"https://syntology.ai/paper/2507.02768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02768"}},"official":{"repos":["kehanlu/desta2.5-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flash-vstream-efficient-real-time","slug":"flash-vstream-efficient-real-time","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","date":"2025-06-30","arxiv_id":"2506.23825","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":0,"n_instrument":7,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/flash-vstream-efficient-real-time#ran","syntology_url":"https://syntology.ai/paper/2506.23825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.23825"}},"official":{"repos":["IVGSZ/Flash-VStream"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tavis-text-bridged-audio-visual-segmentation","slug":"tavis-text-bridged-audio-visual-segmentation","title":"TAViS: Text-bridged Audio-Visual Segmentation with Foundation Models","date":"2025-06-13","arxiv_id":"2506.11436","repositories_listed":0,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tavis-text-bridged-audio-visual-segmentation#ran","syntology_url":"https://syntology.ai/paper/2506.11436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11436"}},"official":null}},{"url":"/paper/modality-curation-building-universal","slug":"modality-curation-building-universal","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","date":"2025-05-26","arxiv_id":"2505.19650","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/modality-curation-building-universal#ran","syntology_url":"https://syntology.ai/paper/2505.19650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19650"}},"official":{"repos":["friedrichor/UNITE"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/msci-addressing-clip-s-inherent-limitations","slug":"msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","arxiv_id":"2505.10289","repositories_listed":1,"syntology":{"n":28,"n_ran":23,"n_constructed":13,"n_ran_checked":19,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":28,"phrase":"23 ran (of which 13 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/msci-addressing-clip-s-inherent-limitations#ran","syntology_url":"https://syntology.ai/paper/2505.10289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10289"}},"official":{"repos":["ltpwy/msci"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":13,"n_ran_no_instrument_failure":19,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","slug":"mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathcoder-vl-bridging-vision-and-code-for#ran","syntology_url":"https://syntology.ai/paper/2505.10557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10557"}},"official":{"repos":["mathllm/mathcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cav-mae-sync-improving-contrastive-audio","slug":"cav-mae-sync-improving-contrastive-audio","title":"CAV-MAE Sync: Improving Contrastive Audio-Visual Mask Autoencoders via Fine-Grained Alignment","date":"2025-05-02","arxiv_id":"2505.01237","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cav-mae-sync-improving-contrastive-audio#ran","syntology_url":"https://syntology.ai/paper/2505.01237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.01237"}},"official":{"repos":["edsonroteia/cav-mae-sync"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lposs-label-propagation-over-patches-and","slug":"lposs-label-propagation-over-patches-and","title":"LPOSS: Label Propagation Over Patches and Pixels for Open-vocabulary Semantic Segmentation","date":"2025-03-25","arxiv_id":"2503.19777","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lposs-label-propagation-over-patches-and#ran","syntology_url":"https://syntology.ai/paper/2503.19777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19777"}},"official":{"repos":["vladan-stojnic/lposs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ov-scan-semantically-consistent-alignment-for","slug":"ov-scan-semantically-consistent-alignment-for","title":"OV-SCAN: Semantically Consistent Alignment for Novel Object Discovery in Open-Vocabulary 3D Object Detection","date":"2025-03-09","arxiv_id":"2503.06435","repositories_listed":0,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ov-scan-semantically-consistent-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2503.06435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06435"}},"official":null}},{"url":"/paper/gem-empowering-mllm-for-grounded-ecg","slug":"gem-empowering-mllm-for-grounded-ecg","title":"GEM: Empowering MLLM for Grounded ECG Understanding with Time Series and Images","date":"2025-03-08","arxiv_id":"2503.06073","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gem-empowering-mllm-for-grounded-ecg#ran","syntology_url":"https://syntology.ai/paper/2503.06073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06073"}},"official":{"repos":["lanxiang1017/gem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-causal-relation-alignment-for-1","slug":"cross-modal-causal-relation-alignment-for-1","title":"Cross-modal Causal Relation Alignment for Video Question Grounding","date":"2025-03-05","arxiv_id":"2503.07635","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":12,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":16,"phrase":"13 ran (of which 12 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cross-modal-causal-relation-alignment-for-1#ran","syntology_url":"https://syntology.ai/paper/2503.07635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07635"}},"official":{"repos":["wissingchen/cra-gqa"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":12,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/phantom-subject-consistent-video-generation","slug":"phantom-subject-consistent-video-generation","title":"Phantom: Subject-consistent video generation via cross-modal alignment","date":"2025-02-16","arxiv_id":"2502.11079","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/phantom-subject-consistent-video-generation#ran","syntology_url":"https://syntology.ai/paper/2502.11079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11079"}},"official":null}},{"url":"/paper/ola-pushing-the-frontiers-of-omni-modal","slug":"ola-pushing-the-frontiers-of-omni-modal","title":"Ola: Pushing the Frontiers of Omni-Modal Language Model","date":"2025-02-06","arxiv_id":"2502.04328","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ola-pushing-the-frontiers-of-omni-modal#ran","syntology_url":"https://syntology.ai/paper/2502.04328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04328"}},"official":{"repos":["ola-omni/ola"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-behaves-like-a-bag-of-words-model-cross","slug":"clip-behaves-like-a-bag-of-words-model-cross","title":"CLIP Behaves like a Bag-of-Words Model Cross-modally but not Uni-modally","date":"2025-02-05","arxiv_id":"2502.03566","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip-behaves-like-a-bag-of-words-model-cross#ran","syntology_url":"https://syntology.ai/paper/2502.03566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03566"}},"official":{"repos":["kdariina/clip-not-bow-unimodally"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/align-kd-distilling-cross-modal-alignment","slug":"align-kd-distilling-cross-modal-alignment","title":"Align-KD: Distilling Cross-Modal Alignment Knowledge for Mobile Vision-Language Model","date":"2024-12-02","arxiv_id":"2412.01282","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/align-kd-distilling-cross-modal-alignment#ran","syntology_url":"https://syntology.ai/paper/2412.01282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01282"}},"official":{"repos":["fqhank/align-kd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-semantic-variation-in-text-to","slug":"evaluating-semantic-variation-in-text-to","title":"Evaluating Semantic Variation in Text-to-Image Synthesis: A Causal Perspective","date":"2024-10-14","arxiv_id":"2410.10291","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-semantic-variation-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2410.10291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10291"}},"official":{"repos":["zhuxiangru/semvarbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deciphering-cross-modal-alignment-in-large","slug":"deciphering-cross-modal-alignment-in-large","title":"Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate","date":"2024-10-09","arxiv_id":"2410.07167","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deciphering-cross-modal-alignment-in-large#ran","syntology_url":"https://syntology.ai/paper/2410.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07167"}},"official":{"repos":["shikiw/modality-integration-rate"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/c-melt-contrastive-enhanced-masked-auto","slug":"c-melt-contrastive-enhanced-masked-auto","title":"Boosting Masked ECG-Text Auto-Encoders as Discriminative Learners","date":"2024-10-03","arxiv_id":"2410.02131","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/c-melt-contrastive-enhanced-masked-auto#ran","syntology_url":"https://syntology.ai/paper/2410.02131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02131"}},"official":{"repos":["manhph2211/d-beta"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mapper-multimodal-prior-guided-parameter","slug":"mapper-multimodal-prior-guided-parameter","title":"MaPPER: Multimodal Prior-guided Parameter Efficient Tuning for Referring Expression Comprehension","date":"2024-09-20","arxiv_id":"2409.13609","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mapper-multimodal-prior-guided-parameter#ran","syntology_url":"https://syntology.ai/paper/2409.13609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13609"}},"official":{"repos":["liuting20/mapper"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-multi-grained-alignment-for","slug":"advancing-multi-grained-alignment-for","title":"Advancing Multi-grained Alignment for Contrastive Language-Audio Pre-training","date":"2024-08-15","arxiv_id":"2408.07919","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/advancing-multi-grained-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2408.07919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07919"}},"official":{"repos":["ming-er/mga-clap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00969","slug":"2408-00969","title":"Visible-Thermal Multiple Object Tracking: Large-scale Video Dataset and Progressive Fusion Approach","date":"2024-08-02","arxiv_id":"2408.00969","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2408-00969#ran","syntology_url":"https://syntology.ai/paper/2408.00969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00969"}},"official":{"repos":["wqw123wqw/pftrack"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigate-the-gap-investigating-approaches-for","slug":"mitigate-the-gap-investigating-approaches-for","title":"Mitigate the Gap: Investigating Approaches for Improving Cross-Modal Alignment in CLIP","date":"2024-06-25","arxiv_id":"2406.17639","repositories_listed":1,"syntology":{"n":22,"n_ran":16,"n_constructed":0,"n_ran_checked":10,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":3,"n_no_contract":7,"n_pointer_only":22,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 3 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mitigate-the-gap-investigating-approaches-for#ran","syntology_url":"https://syntology.ai/paper/2406.17639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17639"}},"official":{"repos":["sarahesl/alignclip"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/presto-progressive-pretraining-enhances","slug":"presto-progressive-pretraining-enhances","title":"PRESTO: Progressive Pretraining Enhances Synthetic Chemistry Outcomes","date":"2024-06-19","arxiv_id":"2406.13193","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/presto-progressive-pretraining-enhances#ran","syntology_url":"https://syntology.ai/paper/2406.13193","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13193"}},"official":{"repos":["idea-xl/presto"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/predicting-polymer-properties-based-on","slug":"predicting-polymer-properties-based-on","title":"MMPolymer: A Multimodal Multitask Pretraining Framework for Polymer Property Prediction","date":"2024-06-07","arxiv_id":"2406.04727","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predicting-polymer-properties-based-on#ran","syntology_url":"https://syntology.ai/paper/2406.04727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04727"}},"official":{"repos":["fanmengwang/mmpolymer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-novel-object-discovery-and-box","slug":"collaborative-novel-object-discovery-and-box","title":"Collaborative Novel Object Discovery and Box-Guided Cross-Modal Alignment for Open-Vocabulary 3D Object Detection","date":"2024-06-02","arxiv_id":"2406.00830","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/collaborative-novel-object-discovery-and-box#ran","syntology_url":"https://syntology.ai/paper/2406.00830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00830"}},"official":{"repos":["yangcaoai/CoDA_NeurIPS2023"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deco-decoupling-token-compression-from","slug":"deco-decoupling-token-compression-from","title":"DeCo: Decoupling Token Compression from Semantic Abstraction in Multimodal Large Language Models","date":"2024-05-31","arxiv_id":"2405.20985","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deco-decoupling-token-compression-from#ran","syntology_url":"https://syntology.ai/paper/2405.20985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20985"}},"official":{"repos":["yaolinli/deco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-feature-adaptation-for-3d","slug":"self-supervised-feature-adaptation-for-3d","title":"Self-supervised Feature Adaptation for 3D Industrial Anomaly Detection","date":"2024-01-06","arxiv_id":"2401.03145","repositories_listed":0,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-feature-adaptation-for-3d#ran","syntology_url":"https://syntology.ai/paper/2401.03145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03145"}},"official":null}},{"url":"/paper/auffusion-leveraging-the-power-of-diffusion","slug":"auffusion-leveraging-the-power-of-diffusion","title":"Auffusion: Leveraging the Power of Diffusion and Large Language Models for Text-to-Audio Generation","date":"2024-01-02","arxiv_id":"2401.01044","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/auffusion-leveraging-the-power-of-diffusion#ran","syntology_url":"https://syntology.ai/paper/2401.01044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01044"}},"official":{"repos":["happylittlecat2333/Auffusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/towards-balanced-alignment-modal-enhanced","slug":"towards-balanced-alignment-modal-enhanced","title":"Towards Balanced Alignment: Modal-Enhanced Semantic Modeling for Video Moment Retrieval","date":"2023-12-19","arxiv_id":"2312.12155","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-balanced-alignment-modal-enhanced#ran","syntology_url":"https://syntology.ai/paper/2312.12155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12155"}},"official":{"repos":["lntzm/mesm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mask-grounding-for-referring-image","slug":"mask-grounding-for-referring-image","title":"Mask Grounding for Referring Image Segmentation","date":"2023-12-19","arxiv_id":"2312.12198","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mask-grounding-for-referring-image#ran","syntology_url":"https://syntology.ai/paper/2312.12198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12198"}},"official":{"repos":["yxchng/mask-grounding"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/navigating-open-set-scenarios-for-skeleton","slug":"navigating-open-set-scenarios-for-skeleton","title":"Navigating Open Set Scenarios for Skeleton-based Action Recognition","date":"2023-12-11","arxiv_id":"2312.06330","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/navigating-open-set-scenarios-for-skeleton#ran","syntology_url":"https://syntology.ai/paper/2312.06330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06330"}},"official":{"repos":["kpeng9510/os-sar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/position-enhanced-visual-instruction-tuning","slug":"position-enhanced-visual-instruction-tuning","title":"Position-Enhanced Visual Instruction Tuning for Multimodal Large Language Models","date":"2023-08-25","arxiv_id":"2308.13437","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/position-enhanced-visual-instruction-tuning#ran","syntology_url":"https://syntology.ai/paper/2308.13437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13437"}},"official":{"repos":["pvit-official/pvit"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-entity-landmark-adaptive-pre","slug":"grounded-entity-landmark-adaptive-pre","title":"Grounded Entity-Landmark Adaptive Pre-training for Vision-and-Language Navigation","date":"2023-08-24","arxiv_id":"2308.12587","repositories_listed":1,"syntology":{"n":30,"n_ran":23,"n_constructed":14,"n_ran_checked":20,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":30,"phrase":"23 ran (of which 14 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/grounded-entity-landmark-adaptive-pre#ran","syntology_url":"https://syntology.ai/paper/2308.12587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12587"}},"official":{"repos":["csir1996/vln-gela"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":14,"n_ran_no_instrument_failure":20,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/aerialvln-vision-and-language-navigation-for","slug":"aerialvln-vision-and-language-navigation-for","title":"AerialVLN: Vision-and-Language Navigation for UAVs","date":"2023-08-13","arxiv_id":"2308.06735","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/aerialvln-vision-and-language-navigation-for#ran","syntology_url":"https://syntology.ai/paper/2308.06735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06735"}},"official":{"repos":["airvln/airvln"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-multimodal-representation-in","slug":"revisiting-multimodal-representation-in","title":"Revisiting Multimodal Representation in Contrastive Learning: From Patch and Token Embeddings to Finite Discrete Tokens","date":"2023-03-27","arxiv_id":"2303.14865","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revisiting-multimodal-representation-in#ran","syntology_url":"https://syntology.ai/paper/2303.14865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14865"}},"official":{"repos":["yuxiaochen1103/fdt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cvt-slr-contrastive-visual-textual","slug":"cvt-slr-contrastive-visual-textual","title":"CVT-SLR: Contrastive Visual-Textual Transformation for Sign Language Recognition with Variational Alignment","date":"2023-03-10","arxiv_id":"2303.05725","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cvt-slr-contrastive-visual-textual#ran","syntology_url":"https://syntology.ai/paper/2303.05725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05725"}},"official":{"repos":["binbinjiang/cvt-slr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-diffusion-learning-multi-modal-diffusion","slug":"mm-diffusion-learning-multi-modal-diffusion","title":"MM-Diffusion: Learning Multi-Modal Diffusion Models for Joint Audio and Video Generation","date":"2022-12-19","arxiv_id":"2212.09478","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 3 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mm-diffusion-learning-multi-modal-diffusion#ran","syntology_url":"https://syntology.ai/paper/2212.09478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09478"}},"official":{"repos":["researchmm/mm-diffusion"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-driven-fine-grained-text-image-person-re","slug":"clip-driven-fine-grained-text-image-person-re","title":"CLIP-Driven Fine-grained Text-Image Person Re-identification","date":"2022-10-19","arxiv_id":"2210.10276","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clip-driven-fine-grained-text-image-person-re#ran","syntology_url":"https://syntology.ai/paper/2210.10276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10276"}},"official":{"repos":["shuanglinyan/CFine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/low-resource-neural-machine-translation-with","slug":"low-resource-neural-machine-translation-with","title":"Low-resource Neural Machine Translation with Cross-modal Alignment","date":"2022-10-13","arxiv_id":"2210.06716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/low-resource-neural-machine-translation-with#ran","syntology_url":"https://syntology.ai/paper/2210.06716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06716"}},"official":{"repos":["ictnlp/lnmt-ca"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-image-descriptions-via-sequential","slug":"generating-image-descriptions-via-sequential","title":"Generating Image Descriptions via Sequential Cross-Modal Alignment Guided by Human Gaze","date":"2020-11-09","arxiv_id":"2011.04592","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/generating-image-descriptions-via-sequential#ran","syntology_url":"https://syntology.ai/paper/2011.04592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.04592"}},"official":{"repos":["dmg-illc/didec-seq-gen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-coupled-autoencoder-approach-for-multi-1","slug":"a-coupled-autoencoder-approach-for-multi-1","title":"A coupled autoencoder approach for multi-modal analysis of cell types","date":"2019-11-06","arxiv_id":"1911.05663","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-coupled-autoencoder-approach-for-multi-1#ran","syntology_url":"https://syntology.ai/paper/1911.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.05663"}},"official":{"repos":["AllenInstitute/coupledAE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"65630c4a0f5d002267300a4eeb5aa84f14121cb4b06da00544b2abbbf852dbee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}