{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-alignment/papers/2","list_of":"/task/cross-modal-alignment","task":"cross-modal alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":342,"counts":{"archive_papers_tagged":342,"with_a_code_link":151,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":342,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":41,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":41,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-alignment","prev":"/task/cross-modal-alignment","next":"/task/cross-modal-alignment/papers/3","papers":[{"url":"/paper/mmdesign-multi-modality-transfer-learning-for","slug":"mmdesign-multi-modality-transfer-learning-for","title":"Progressive Multi-Modality Learning for Inverse Protein Folding","date":"2023-12-11","arxiv_id":"2312.06297","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-open-set-scenarios-for-skeleton","slug":"navigating-open-set-scenarios-for-skeleton","title":"Navigating Open Set Scenarios for Skeleton-based Action Recognition","date":"2023-12-11","arxiv_id":"2312.06330","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/navigating-open-set-scenarios-for-skeleton#ran","syntology_url":"https://syntology.ai/paper/2312.06330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06330"}},"official":{"repos":["kpeng9510/os-sar"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/coda-collaborative-novel-box-discovery-and-1","slug":"coda-collaborative-novel-box-discovery-and-1","title":"CoDA: Collaborative Novel Box Discovery and Cross-modal Alignment for Open-vocabulary 3D Object Detection","date":"2023-10-04","arxiv_id":"2310.02960","repositories_listed":1,"syntology":null},{"url":"/paper/reform-eval-evaluating-large-vision-language","slug":"reform-eval-evaluating-large-vision-language","title":"ReForm-Eval: Evaluating Large Vision Language Models via Unified Re-Formulation of Task-Oriented Benchmarks","date":"2023-10-04","arxiv_id":"2310.02569","repositories_listed":1,"syntology":null},{"url":"/paper/align-before-search-aligning-ads-image-to","slug":"align-before-search-aligning-ads-image-to","title":"Align before Search: Aligning Ads Image to Text for Accurate Cross-Modal Sponsored Search","date":"2023-09-28","arxiv_id":"2309.16141","repositories_listed":1,"syntology":null},{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-semantic-fusion-model-for-generalized","slug":"multi-semantic-fusion-model-for-generalized","title":"Multi-Semantic Fusion Model for Generalized Zero-Shot Skeleton-Based Action Recognition","date":"2023-09-18","arxiv_id":"2309.09592","repositories_listed":1,"syntology":null},{"url":"/paper/position-enhanced-visual-instruction-tuning","slug":"position-enhanced-visual-instruction-tuning","title":"Position-Enhanced Visual Instruction Tuning for Multimodal Large Language Models","date":"2023-08-25","arxiv_id":"2308.13437","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/position-enhanced-visual-instruction-tuning#ran","syntology_url":"https://syntology.ai/paper/2308.13437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13437"}},"official":{"repos":["pvit-official/pvit"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-entity-landmark-adaptive-pre","slug":"grounded-entity-landmark-adaptive-pre","title":"Grounded Entity-Landmark Adaptive Pre-training for Vision-and-Language Navigation","date":"2023-08-24","arxiv_id":"2308.12587","repositories_listed":1,"syntology":{"n":30,"n_ran":23,"n_constructed":14,"n_ran_checked":20,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":30,"phrase":"23 ran (of which 14 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/grounded-entity-landmark-adaptive-pre#ran","syntology_url":"https://syntology.ai/paper/2308.12587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12587"}},"official":{"repos":["csir1996/vln-gela"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":14,"n_ran_no_instrument_failure":20,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/language-guided-diffusion-model-for-visual","slug":"language-guided-diffusion-model-for-visual","title":"Language-Guided Diffusion Model for Visual Grounding","date":"2023-08-18","arxiv_id":"2308.09599","repositories_listed":1,"syntology":null},{"url":"/paper/aerialvln-vision-and-language-navigation-for","slug":"aerialvln-vision-and-language-navigation-for","title":"AerialVLN: Vision-and-Language Navigation for UAVs","date":"2023-08-13","arxiv_id":"2308.06735","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/aerialvln-vision-and-language-navigation-for#ran","syntology_url":"https://syntology.ai/paper/2308.06735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06735"}},"official":{"repos":["airvln/airvln"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/contrast-augmented-diffusion-model-with-fine","slug":"contrast-augmented-diffusion-model-with-fine","title":"Contrast-augmented Diffusion Model with Fine-grained Sequence Alignment for Markup-to-Image Generation","date":"2023-08-02","arxiv_id":"2308.01147","repositories_listed":1,"syntology":null},{"url":"/paper/unleashing-the-imagination-of-text-a-novel","slug":"unleashing-the-imagination-of-text-a-novel","title":"Text-guided Image Restoration and Semantic Enhancement for Text-to-Image Person Retrieval","date":"2023-07-18","arxiv_id":"2307.09059","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-semantic-completion-learning","slug":"global-and-local-semantic-completion-learning","title":"Global and Local Semantic Completion Learning for Vision-Language Pre-training","date":"2023-06-12","arxiv_id":"2306.07096","repositories_listed":1,"syntology":null},{"url":"/paper/managertower-aggregating-the-insights-of-uni","slug":"managertower-aggregating-the-insights-of-uni","title":"ManagerTower: Aggregating the Insights of Uni-Modal Experts for Vision-Language Representation Learning","date":"2023-05-31","arxiv_id":"2306.00103","repositories_listed":1,"syntology":null},{"url":"/paper/soc-semantic-assisted-object-cluster-for","slug":"soc-semantic-assisted-object-cluster-for","title":"SOC: Semantic-Assisted Object Cluster for Referring Video Object Segmentation","date":"2023-05-26","arxiv_id":"2305.17011","repositories_listed":1,"syntology":null},{"url":"/paper/speech-text-dialog-pre-training-for-spoken","slug":"speech-text-dialog-pre-training-for-spoken","title":"Speech-Text Dialog Pre-training for Spoken Dialog Understanding with Explicit Cross-Modal Alignment","date":"2023-05-19","arxiv_id":"2305.11579","repositories_listed":1,"syntology":null},{"url":"/paper/towards-medical-artificial-general","slug":"towards-medical-artificial-general","title":"Towards Medical Artificial General Intelligence via Knowledge-Enhanced Multimodal Pretraining","date":"2023-04-26","arxiv_id":"2304.14204","repositories_listed":1,"syntology":null},{"url":"/paper/a-closer-look-at-audio-visual-semantic","slug":"a-closer-look-at-audio-visual-semantic","title":"Unraveling Instance Associations: A Closer Look for Audio-Visual Segmentation","date":"2023-04-06","arxiv_id":"2304.02970","repositories_listed":1,"syntology":null},{"url":"/paper/unmasked-teacher-towards-training-efficient","slug":"unmasked-teacher-towards-training-efficient","title":"Unmasked Teacher: Towards Training-Efficient Video Foundation Models","date":"2023-03-28","arxiv_id":"2303.16058","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unmasked-teacher-towards-training-efficient#ran","syntology_url":"https://syntology.ai/paper/2303.16058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16058"}},"official":{"repos":["opengvlab/unmasked_teacher"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-multimodal-representation-in","slug":"revisiting-multimodal-representation-in","title":"Revisiting Multimodal Representation in Contrastive Learning: From Patch and Token Embeddings to Finite Discrete Tokens","date":"2023-03-27","arxiv_id":"2303.14865","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/revisiting-multimodal-representation-in#ran","syntology_url":"https://syntology.ai/paper/2303.14865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14865"}},"official":{"repos":["yuxiaochen1103/fdt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cvt-slr-contrastive-visual-textual","slug":"cvt-slr-contrastive-visual-textual","title":"CVT-SLR: Contrastive Visual-Textual Transformation for Sign Language Recognition with Variational Alignment","date":"2023-03-10","arxiv_id":"2303.05725","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cvt-slr-contrastive-visual-textual#ran","syntology_url":"https://syntology.ai/paper/2303.05725","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05725"}},"official":{"repos":["binbinjiang/cvt-slr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/logonet-towards-accurate-3d-object-detection","slug":"logonet-towards-accurate-3d-object-detection","title":"LoGoNet: Towards Accurate 3D Object Detection with Local-to-Global Cross-Modal Fusion","date":"2023-03-07","arxiv_id":"2303.03595","repositories_listed":1,"syntology":null},{"url":"/paper/hiclip-contrastive-language-image-pretraining","slug":"hiclip-contrastive-language-image-pretraining","title":"HiCLIP: Contrastive Language-Image Pretraining with Hierarchy-aware Attention","date":"2023-03-06","arxiv_id":"2303.02995","repositories_listed":1,"syntology":null},{"url":"/paper/mm-diffusion-learning-multi-modal-diffusion","slug":"mm-diffusion-learning-multi-modal-diffusion","title":"MM-Diffusion: Learning Multi-Modal Diffusion Models for Joint Audio and Video Generation","date":"2022-12-19","arxiv_id":"2212.09478","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":12,"n_instrument":4,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":9,"n_pointer_only":8,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 3 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mm-diffusion-learning-multi-modal-diffusion#ran","syntology_url":"https://syntology.ai/paper/2212.09478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09478"}},"official":{"repos":["researchmm/mm-diffusion"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/asymmetric-cross-scale-alignment-for-text","slug":"asymmetric-cross-scale-alignment-for-text","title":"Asymmetric Cross-Scale Alignment for Text-Based Person Search","date":"2022-11-26","arxiv_id":"2212.11958","repositories_listed":1,"syntology":null},{"url":"/paper/seeing-what-you-miss-vision-language-pre","slug":"seeing-what-you-miss-vision-language-pre","title":"Seeing What You Miss: Vision-Language Pre-training with Semantic Completion Learning","date":"2022-11-24","arxiv_id":"2211.13437","repositories_listed":1,"syntology":null},{"url":"/paper/camanet-class-activation-map-guided-attention","slug":"camanet-class-activation-map-guided-attention","title":"CAMANet: Class Activation Map Guided Attention Network for Radiology Report Generation","date":"2022-11-02","arxiv_id":"2211.01412","repositories_listed":1,"syntology":null},{"url":"/paper/clip-driven-fine-grained-text-image-person-re","slug":"clip-driven-fine-grained-text-image-person-re","title":"CLIP-Driven Fine-grained Text-Image Person Re-identification","date":"2022-10-19","arxiv_id":"2210.10276","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":2,"n_instrument":8,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clip-driven-fine-grained-text-image-person-re#ran","syntology_url":"https://syntology.ai/paper/2210.10276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10276"}},"official":{"repos":["shuanglinyan/CFine"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/discrete-cross-modal-alignment-enables-zero","slug":"discrete-cross-modal-alignment-enables-zero","title":"Discrete Cross-Modal Alignment Enables Zero-Shot Speech Translation","date":"2022-10-18","arxiv_id":"2210.09556","repositories_listed":1,"syntology":null},{"url":"/paper/low-resource-neural-machine-translation-with","slug":"low-resource-neural-machine-translation-with","title":"Low-resource Neural Machine Translation with Cross-modal Alignment","date":"2022-10-13","arxiv_id":"2210.06716","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/low-resource-neural-machine-translation-with#ran","syntology_url":"https://syntology.ai/paper/2210.06716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06716"}},"official":{"repos":["ictnlp/lnmt-ca"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-semantically-aligned-vision","slug":"fine-grained-semantically-aligned-vision","title":"Fine-Grained Semantically Aligned Vision-Language Pre-Training","date":"2022-08-04","arxiv_id":"2208.02515","repositories_listed":1,"syntology":null},{"url":"/paper/a-priority-map-for-vision-and-language","slug":"a-priority-map-for-vision-and-language","title":"A Priority Map for Vision-and-Language Navigation with Trajectory Plans and Feature-Location Cues","date":"2022-07-24","arxiv_id":"2207.11717","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-cross-modal-alignment-for","slug":"reinforced-cross-modal-alignment-for","title":"Reinforced Cross-modal Alignment for Radiology Report Generation","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dsgn-exploiting-visual-spatial-relation","slug":"dsgn-exploiting-visual-spatial-relation","title":"DSGN++: Exploiting Visual-Spatial Relation for Stereo-based 3D Detectors","date":"2022-04-06","arxiv_id":"2204.03039","repositories_listed":1,"syntology":null},{"url":"/paper/learning-commonsense-aware-moment-text","slug":"learning-commonsense-aware-moment-text","title":"Learning Commonsense-aware Moment-Text Alignment for Fast Video Temporal Grounding","date":"2022-04-04","arxiv_id":"2204.01450","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-multi","slug":"ernie-layout-layout-knowledge-enhanced-multi","title":"ERNIE-Layout: Layout-Knowledge Enhanced Multi-modal Pre-training for Document Understanding","date":"2022-01-16","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/align-and-prompt-video-and-language-pre","slug":"align-and-prompt-video-and-language-pre","title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts","date":"2021-12-17","arxiv_id":"2112.09583","repositories_listed":1,"syntology":null},{"url":"/paper/landmark-rxr-solving-vision-and-language","slug":"landmark-rxr-solving-vision-and-language","title":"Landmark-RxR: Solving Vision-and-Language Navigation with Fine-Grained Alignment Supervision","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/kd-vlp-improving-end-to-end-vision-and","slug":"kd-vlp-improving-end-to-end-vision-and","title":"KD-VLP: Improving End-to-End Vision-and-Language Pretraining with Object Knowledge Distillation","date":"2021-09-22","arxiv_id":"2109.10504","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-modality-interaction-modeling-for","slug":"dynamic-modality-interaction-modeling-for","title":"Dynamic Modality Interaction Modeling for Image-Text Retrieval","date":"2021-07-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/perception-aware-multi-sensor-fusion-for-3d","slug":"perception-aware-multi-sensor-fusion-for-3d","title":"EPMF: Efficient Perception-aware Multi-sensor Fusion for 3D Semantic Segmentation","date":"2021-06-21","arxiv_id":"2106.15277","repositories_listed":1,"syntology":null},{"url":"/paper/improving-cross-modal-alignment-in-vision","slug":"improving-cross-modal-alignment-in-vision","title":"Improving Cross-Modal Alignment in Vision Language Navigation via Syntactic Information","date":"2021-04-19","arxiv_id":"2104.09580","repositories_listed":1,"syntology":null},{"url":"/paper/generating-image-descriptions-via-sequential","slug":"generating-image-descriptions-via-sequential","title":"Generating Image Descriptions via Sequential Cross-Modal Alignment Guided by Human Gaze","date":"2020-11-09","arxiv_id":"2011.04592","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/generating-image-descriptions-via-sequential#ran","syntology_url":"https://syntology.ai/paper/2011.04592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.04592"}},"official":{"repos":["dmg-illc/didec-seq-gen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-contextual-tag-embeddings-for-cross","slug":"learning-contextual-tag-embeddings-for-cross","title":"Learning Contextual Tag Embeddings for Cross-Modal Alignment of Audio and Tags","date":"2020-10-27","arxiv_id":"2010.14171","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-contextual-tag-embeddings-for-cross#ran","syntology_url":"https://syntology.ai/paper/2010.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14171"}},"official":{"repos":["xavierfav/ae-w2v-attention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/danceit-music-inspired-dancing-video","slug":"danceit-music-inspired-dancing-video","title":"DanceIt: Music-inspired Dancing Video Synthesis","date":"2020-09-17","arxiv_id":"2009.08027","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-cross-modal-alignment-for-multi","slug":"unsupervised-cross-modal-alignment-for-multi","title":"Unsupervised Cross-Modal Alignment for Multi-Person 3D Pose Estimation","date":"2020-08-04","arxiv_id":"2008.01388","repositories_listed":1,"syntology":null},{"url":"/paper/symbiotic-adversarial-learning-for-attribute","slug":"symbiotic-adversarial-learning-for-attribute","title":"Symbiotic Adversarial Learning for Attribute-based Person Search","date":"2020-07-19","arxiv_id":"2007.09609","repositories_listed":1,"syntology":null},{"url":"/paper/a-coupled-autoencoder-approach-for-multi-1","slug":"a-coupled-autoencoder-approach-for-multi-1","title":"A coupled autoencoder approach for multi-modal analysis of cell types","date":"2019-11-06","arxiv_id":"1911.05663","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-coupled-autoencoder-approach-for-multi-1#ran","syntology_url":"https://syntology.ai/paper/1911.05663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.05663"}},"official":{"repos":["AllenInstitute/coupledAE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":null,"slug":"transformer-based-spatial-grounding-a","title":"Transformer-based Spatial Grounding: A Comprehensive Survey","date":"2025-07-17","arxiv_id":"2507.12739","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridge-feature-matching-and-cross-modal","title":"Bridge Feature Matching and Cross-Modal Alignment with Mutual-filtering for Zero-shot Anomaly Detection","date":"2025-07-15","arxiv_id":"2507.11003","repositories_listed":0,"syntology":null},{"url":null,"slug":"catvis-context-aware-thought-visualization","title":"CATVis: Context-Aware Thought Visualization","date":"2025-07-15","arxiv_id":"2507.11522","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-attribute-confusion-in-fashion","title":"Evaluating Attribute Confusion in Fashion Text-to-Image Generation","date":"2025-07-09","arxiv_id":"2507.07079","repositories_listed":0,"syntology":null},{"url":null,"slug":"dalr-dual-level-alignment-learning-for","title":"DALR: Dual-level Alignment Learning for Multimodal Sentence Representation Learning","date":"2025-06-26","arxiv_id":"2506.21096","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsdaseg-a-two-stage-model-with-direct","title":"TSDASeg: A Two-Stage Model with Direct Alignment for Interactive Point Cloud Segmentation","date":"2025-06-26","arxiv_id":"2506.20991","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-language-models-with-decoupled","title":"Speech-Language Models with Decoupled Tokenizers and Multi-Token Prediction","date":"2025-06-14","arxiv_id":"2506.12537","repositories_listed":0,"syntology":null},{"url":"/paper/tavis-text-bridged-audio-visual-segmentation","slug":"tavis-text-bridged-audio-visual-segmentation","title":"TAViS: Text-bridged Audio-Visual Segmentation with Foundation Models","date":"2025-06-13","arxiv_id":"2506.11436","repositories_listed":0,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tavis-text-bridged-audio-visual-segmentation#ran","syntology_url":"https://syntology.ai/paper/2506.11436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11436"}},"official":null}},{"url":null,"slug":"improving-medical-visual-representation","title":"Improving Medical Visual Representation Learning with Pathological-level Cross-Modal Alignment and Correlation Exploration","date":"2025-06-12","arxiv_id":"2506.10573","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-cross-modal-and-uni-modal","title":"Fusing Cross-modal and Uni-modal Representations: A Kronecker Product Approach","date":"2025-06-10","arxiv_id":"2506.08645","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-vision-language-navigation","title":"Generating Vision-Language Navigation Instructions Incorporated Fine-Grained Alignment Annotations","date":"2025-06-10","arxiv_id":"2506.08566","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisq-cross-modal-representation-learning-for","title":"WhisQ: Cross-Modal Representation Learning for Text-to-Music MOS Prediction","date":"2025-06-06","arxiv_id":"2506.05899","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-your-3d-encoder-really-work-when","title":"Does Your 3D Encoder Really Work? When Pretrain-SFT from 2D VLMs Meets 3D VLMs","date":"2025-06-05","arxiv_id":"2506.05318","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-llm-centric-multimodal-fusion-a-1","title":"Towards LLM-Centric Multimodal Fusion: A Survey on Integration Strategies and Techniques","date":"2025-06-05","arxiv_id":"2506.04788","repositories_listed":0,"syntology":null},{"url":null,"slug":"unicue-unified-recognition-and-generation","title":"UniCUE: Unified Recognition and Generation Framework for Chinese Cued Speech Video-to-Speech Generation","date":"2025-06-04","arxiv_id":"2506.04134","repositories_listed":0,"syntology":null},{"url":null,"slug":"emotionrankclap-bridging-natural-language","title":"EmotionRankCLAP: Bridging Natural Language Speaking Styles and Ordinal Speech Emotion via Rank-N-Contrast","date":"2025-05-29","arxiv_id":"2505.23732","repositories_listed":0,"syntology":null},{"url":null,"slug":"alas-measuring-latent-speech-text-alignment","title":"ALAS: Measuring Latent Speech-Text Alignment For Spoken Language Understanding In Multimodal LLMs","date":"2025-05-26","arxiv_id":"2505.19937","repositories_listed":0,"syntology":null},{"url":null,"slug":"disa-directional-saliency-aware-prompt","title":"DiSa: Directional Saliency-Aware Prompt Learning for Generalizable Vision-Language Models","date":"2025-05-26","arxiv_id":"2505.19373","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-alignment-to-advancement-bootstrapping","title":"From Alignment to Advancement: Bootstrapping Audio-Language Alignment with Synthetic Data","date":"2025-05-26","arxiv_id":"2505.20166","repositories_listed":0,"syntology":null},{"url":null,"slug":"vitapes-visuotactile-position-encodings-for","title":"ViTaPEs: Visuotactile Position Encodings for Cross-Modal Alignment in Multimodal Transformers","date":"2025-05-26","arxiv_id":"2505.20032","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-attendwg-co-attentive-dimension-wise","title":"Co-AttenDWG: Co-Attentive Dimension-Wise Gating and Expert Fusion for Multi-Modal Offensive Content Detection","date":"2025-05-25","arxiv_id":"2505.19010","repositories_listed":0,"syntology":null},{"url":null,"slug":"deformable-attentive-visual-enhancement-for","title":"Deformable Attentive Visual Enhancement for Referring Segmentation Using Vision-Language Model","date":"2025-05-25","arxiv_id":"2505.19242","repositories_listed":0,"syntology":null},{"url":null,"slug":"mllms-are-deeply-affected-by-modality-bias","title":"MLLMs are Deeply Affected by Modality Bias","date":"2025-05-24","arxiv_id":"2505.18657","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip4retrofit-enabling-real-time-image","title":"Clip4Retrofit: Enabling Real-Time Image Labeling on Edge Devices via Cross-Architecture CLIP Distillation","date":"2025-05-23","arxiv_id":"2505.18039","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-discrepancy-bridging-method","title":"Representation Discrepancy Bridging Method for Remote Sensing Image-Text Retrieval","date":"2025-05-22","arxiv_id":"2505.16756","repositories_listed":0,"syntology":null},{"url":null,"slug":"aln-p3-unified-language-alignment-for","title":"ALN-P3: Unified Language Alignment for Perception, Prediction, and Planning in Autonomous Driving","date":"2025-05-21","arxiv_id":"2505.15158","repositories_listed":0,"syntology":null},{"url":null,"slug":"cad-a-general-multimodal-framework-for-video","title":"CAD: A General Multimodal Framework for Video Deepfake Detection via Cross-Modal Alignment and Distillation","date":"2025-05-21","arxiv_id":"2505.15233","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llms-for-time-series-forecasting","title":"Enhancing LLMs for Time Series Forecasting via Structure-Guided Cross-Modal Alignment","date":"2025-05-19","arxiv_id":"2505.13175","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10917","title":"VISTA: Enhancing Vision-Text Alignment in MLLMs via Cross-Modal Mutual Information Maximization","date":"2025-05-16","arxiv_id":"2505.10917","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11192","title":"FALCON: False-Negative Aware Learning of Contrastive Negatives in Vision-Language Pretraining","date":"2025-05-16","arxiv_id":"2505.11192","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-modality-collapse-representations","title":"Beyond Modality Collapse: Representations Blending for Multimodal Dataset Distillation","date":"2025-05-16","arxiv_id":"2505.14705","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10729","title":"Adaptive Spatial Transcriptomics Interpolation via Cross-modal Cross-slice Modeling","date":"2025-05-15","arxiv_id":"2505.10729","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-and-alignment-rethinking-domain","title":"Denoising and Alignment: Rethinking Domain Generalization for Multimodal Face Anti-Spoofing","date":"2025-05-14","arxiv_id":"2505.09484","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-space-intervened-diffusive-alignment","title":"Semantic-Space-Intervened Diffusive Alignment for Visual Classification","date":"2025-05-09","arxiv_id":"2505.05721","repositories_listed":0,"syntology":null},{"url":null,"slug":"densegrounding-improving-dense-language","title":"DenseGrounding: Improving Dense Language-Vision Semantics for Ego-Centric 3D Visual Grounding","date":"2025-05-08","arxiv_id":"2505.04965","repositories_listed":0,"syntology":null},{"url":null,"slug":"physllm-harnessing-large-language-models-for","title":"PhysLLM: Harnessing Large Language Models for Cross-Modal Remote Physiological Sensing","date":"2025-05-06","arxiv_id":"2505.03621","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-agent-framework-for-automated-1","title":"A Multi-Agent Framework for Automated Qinqiang Opera Script Generation Using Large Language Models","date":"2025-04-22","arxiv_id":"2504.15552","repositories_listed":0,"syntology":null},{"url":"/paper/tmcir-token-merge-benefits-composed-image","slug":"tmcir-token-merge-benefits-composed-image","title":"TMCIR: Token Merge Benefits Composed Image Retrieval","date":"2025-04-15","arxiv_id":"2504.10995","repositories_listed":0,"syntology":null},{"url":null,"slug":"infomae-pair-efficient-cross-modal-alignment","title":"InfoMAE: Pair-Efficient Cross-Modal Alignment for Multimodal Time-Series Sensing Signals","date":"2025-04-13","arxiv_id":"2504.09707","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmt-vision-language-multimodal-transformer","title":"VLMT: Vision-Language Multimodal Transformer for Multimodal Multi-hop Question Answering","date":"2025-04-11","arxiv_id":"2504.08269","repositories_listed":0,"syntology":null},{"url":null,"slug":"se4lip-speech-lip-encoder-for-talking-head","title":"SE4Lip: Speech-Lip Encoder for Talking Head Synthesis to Solve Phoneme-Viseme Alignment Ambiguity","date":"2025-04-08","arxiv_id":"2504.05803","repositories_listed":0,"syntology":null},{"url":null,"slug":"df-calib-targetless-lidar-camera-calibration","title":"DF-Calib: Targetless LiDAR-Camera Calibration via Depth Flow","date":"2025-04-02","arxiv_id":"2504.01416","repositories_listed":0,"syntology":null},{"url":null,"slug":"finelip-extending-clip-s-reach-via-fine","title":"FineLIP: Extending CLIP's Reach via Fine-Grained Alignment with Longer Text Inputs","date":"2025-04-02","arxiv_id":"2504.01916","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-modality-tags-for-enhanced-cross","title":"Leveraging Modality Tags for Enhanced Cross-Modal Video Retrieval","date":"2025-04-02","arxiv_id":"2504.01591","repositories_listed":0,"syntology":null},{"url":null,"slug":"sviqa-a-unified-speech-vision-multimodal","title":"SViQA: A Unified Speech-Vision Multimodal Model for Textless Visual Question Answering","date":"2025-04-01","arxiv_id":"2504.01049","repositories_listed":0,"syntology":null},{"url":null,"slug":"cadformer-fine-grained-cross-modal-alignment","title":"CADFormer: Fine-Grained Cross-modal Alignment and Decoding Transformer for Referring Remote Sensing Image Segmentation","date":"2025-03-30","arxiv_id":"2503.23456","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurolip-interpretable-and-fair-cross-modal","title":"NeuroLIP: Interpretable and Fair Cross-Modal Alignment of fMRI and Phenotypic Text","date":"2025-03-27","arxiv_id":"2503.21964","repositories_listed":0,"syntology":null},{"url":null,"slug":"autorad-lung-a-radiomic-guided-prompting","title":"AutoRad-Lung: A Radiomic-Guided Prompting Autoregressive Vision-Language Model for Lung Nodule Malignancy Prediction","date":"2025-03-26","arxiv_id":"2503.20662","repositories_listed":0,"syntology":null},{"url":null,"slug":"gatedxlstm-a-multimodal-affective-computing","title":"GatedxLSTM: A Multimodal Affective Computing Approach for Emotion Recognition in Conversations","date":"2025-03-26","arxiv_id":"2503.20919","repositories_listed":0,"syntology":null}],"record_sha256":"faeba62595fc7b09ecc424aa6162d72d683ea29407271631989d15e86fe16ed4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}