{"url":"/task/cross-modal-alignment","name":"cross-modal alignment","slug":"cross-modal-alignment","description_markdown":null,"categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":342,"papers_with_code":151,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/gaze-cifar-10","name":"Gaze-CIFAR-10","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":151,"tagged_in_all":342,"items":[{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/mplug-effective-and-efficient-vision-language","title":"mPLUG: Effective and Efficient Vision-Language Learning by Cross-modal Skip-connections","date":"2022-05-24","arxiv_id":"2205.12005","repositories_listed":3,"syntology":null},{"url":"/paper/skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","arxiv_id":"2507.06167","repositories_listed":2,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":12}},{"url":"/paper/desta2-5-audio-toward-general-purpose-large","title":"DeSTA2.5-Audio: Toward General-Purpose Large Audio Language Model with Self-Generated Cross-Modal Alignment","date":"2025-07-03","arxiv_id":"2507.02768","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/multi-granularity-cross-modal-alignment-for","title":"Multi-Granularity Cross-modal Alignment for Generalized Medical Visual Representation Learning","date":"2022-10-12","arxiv_id":"2210.06044","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/bridge-tower-building-bridges-between","title":"BridgeTower: Building Bridges Between Encoders in Vision-Language Representation Learning","date":"2022-06-17","arxiv_id":"2206.08657","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/rsrefseg-2-decoupling-referring-remote","title":"RSRefSeg 2: Decoupling Referring Remote Sensing Image Segmentation with Foundation Models","date":"2025-07-08","arxiv_id":"2507.06231","repositories_listed":1,"syntology":null},{"url":"/paper/flash-vstream-efficient-real-time","title":"Flash-VStream: Efficient Real-Time Understanding for Long Video Streams","date":"2025-06-30","arxiv_id":"2506.23825","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/hyperpath-knowledge-guided-hyperbolic","title":"HyperPath: Knowledge-Guided Hyperbolic Semantic Hierarchy Modeling for WSI Analysis","date":"2025-06-19","arxiv_id":"2506.16398","repositories_listed":1,"syntology":null},{"url":"/paper/reid5o-achieving-omni-multi-modal-person-re","title":"ReID5o: Achieving Omni Multi-modal Person Re-identification in a Single Model","date":"2025-06-11","arxiv_id":"2506.09385","repositories_listed":1,"syntology":null},{"url":"/paper/omnidrca-parallel-speech-text-foundation","title":"OmniDRCA: Parallel Speech-Text Foundation Model via Dual-Resolution Speech Representations and Contrastive Alignment","date":"2025-06-11","arxiv_id":"2506.09349","repositories_listed":1,"syntology":null},{"url":"/paper/modality-curation-building-universal","title":"Modality Curation: Building Universal Embeddings for Advanced Multimodal Information Retrieval","date":"2025-05-26","arxiv_id":"2505.19650","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/icpl-reid-identity-conditional-prompt","title":"ICPL-ReID: Identity-Conditional Prompt Learning for Multi-Spectral Object Re-Identification","date":"2025-05-23","arxiv_id":"2505.17821","repositories_listed":1,"syntology":null},{"url":"/paper/u-sam-an-audio-language-model-for-unified","title":"U-SAM: An audio language Model for Unified Speech, Audio, and Music Understanding","date":"2025-05-20","arxiv_id":"2505.13880","repositories_listed":1,"syntology":null},{"url":"/paper/mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","arxiv_id":"2505.10557","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/msci-addressing-clip-s-inherent-limitations","title":"MSCI: Addressing CLIP's Inherent Limitations for Compositional Zero-Shot Learning","date":"2025-05-15","arxiv_id":"2505.10289","repositories_listed":1,"syntology":{"n":28,"n_ran":23,"n_unverified":5,"n_pointer_only":28}},{"url":"/paper/anatomical-attention-alignment-representation","title":"Anatomical Attention Alignment representation for Radiology Report Generation","date":"2025-05-12","arxiv_id":"2505.07689","repositories_listed":1,"syntology":null},{"url":"/paper/hcma-hierarchical-cross-model-alignment-for","title":"HCMA: Hierarchical Cross-model Alignment for Grounded Text-to-Image Generation","date":"2025-05-10","arxiv_id":"2505.06512","repositories_listed":1,"syntology":null},{"url":"/paper/task-adapter-task-specific-adaptation-with","title":"Task-Adapter++: Task-specific Adaptation with Order-aware Alignment for Few-shot Action Recognition","date":"2025-05-09","arxiv_id":"2505.06002","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-embeddings-for-frozen-vision","title":"Probabilistic Embeddings for Frozen Vision-Language Models: Uncertainty Quantification with Gaussian Process Latent Variable Models","date":"2025-05-08","arxiv_id":"2505.05163","repositories_listed":1,"syntology":null},{"url":"/paper/cav-mae-sync-improving-contrastive-audio","title":"CAV-MAE Sync: Improving Contrastive Audio-Visual Mask Autoencoders via Fine-Grained Alignment","date":"2025-05-02","arxiv_id":"2505.01237","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/micarvlmoe-a-modern-gated-cross-aligned","title":"MicarVLMoE: A Modern Gated Cross-Aligned Vision-Language Mixture of Experts Model for Medical Image Captioning and Report Generation","date":"2025-04-29","arxiv_id":"2504.20343","repositories_listed":1,"syntology":null},{"url":"/paper/cross-attention-for-state-based-model-rwkv-7","title":"Cross-attention for State-based model RWKV-7","date":"2025-04-19","arxiv_id":"2504.14260","repositories_listed":1,"syntology":null},{"url":"/paper/3d-coca-contrastive-learners-are-3d","title":"3D CoCa: Contrastive Learners are 3D Captioners","date":"2025-04-13","arxiv_id":"2504.09518","repositories_listed":1,"syntology":null},{"url":"/paper/gaze-guided-learning-avoiding-shortcut-bias","title":"Gaze-Guided Learning: Avoiding Shortcut Bias in Visual Classification","date":"2025-04-08","arxiv_id":"2504.05583","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-fusion-and-vision-language-models","title":"Multimodal Fusion and Vision-Language Models: A Survey for Robot Vision","date":"2025-04-03","arxiv_id":"2504.02477","repositories_listed":1,"syntology":null},{"url":"/paper/cost-contrastive-one-stage-transformer-for","title":"COST: Contrastive One-Stage Transformer for Vision-Language Small Object Tracking","date":"2025-04-02","arxiv_id":"2504.01321","repositories_listed":1,"syntology":null},{"url":"/paper/bipvl-seg-bidirectional-progressive-vision","title":"BiPVL-Seg: Bidirectional Progressive Vision-Language Fusion with Global-Local Alignment for Medical Image Segmentation","date":"2025-03-30","arxiv_id":"2503.23534","repositories_listed":1,"syntology":null},{"url":"/paper/lposs-label-propagation-over-patches-and","title":"LPOSS: Label Propagation Over Patches and Pixels for Open-vocabulary Semantic Segmentation","date":"2025-03-25","arxiv_id":"2503.19777","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/enhanced-ood-detection-through-cross-modal","title":"Enhanced OoD Detection through Cross-Modal Alignment of Multi-Modal Representations","date":"2025-03-24","arxiv_id":"2503.18817","repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}