{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/cross-modal-alignment/papers/4","list_of":"/task/cross-modal-alignment","task":"cross-modal alignment","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,342],"of":342,"counts":{"archive_papers_tagged":342,"with_a_code_link":151,"where_syntology_ran_a_sample":47,"not_listed_spam_title":0,"listed":342,"listed_where_code_ran":47,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":41,"every_run_a_failure_of_syntologys_instrument":6,"listed_with_a_run_with_no_instrument_failure":41,"listed_every_run_a_failure_of_syntologys_instrument":6,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/cross-modal-alignment","prev":"/task/cross-modal-alignment/papers/3","next":null,"papers":[{"url":null,"slug":"softclip-softer-cross-modal-alignment-makes","title":"SoftCLIP: Softer Cross-modal Alignment Makes CLIP Stronger","date":"2023-03-30","arxiv_id":"2303.17561","repositories_listed":0,"syntology":null},{"url":null,"slug":"tot-topology-aware-optimal-transport-for","title":"TOT: Topology-Aware Optimal Transport For Multimodal Hate Detection","date":"2023-02-27","arxiv_id":"2303.09314","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-semantic-object-detection-with","title":"End-to-end Semantic Object Detection with Cross-Modal Alignment","date":"2023-02-10","arxiv_id":"2302.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-vision-accelerate-hierarchical","title":"Does Vision Accelerate Hierarchical Generalization in Neural Language Learners?","date":"2023-02-01","arxiv_id":"2302.00667","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-modal-alignment-for-text","title":"Improving Cross-modal Alignment for Text-Guided Image Inpainting","date":"2023-01-26","arxiv_id":"2301.11362","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-aligned-cross-modal-representations-1","title":"Linguistic Query-Guided Mask Generation for Referring Image Segmentation","date":"2023-01-16","arxiv_id":"2301.06429","repositories_listed":0,"syntology":null},{"url":"/paper/hitea-hierarchical-temporal-aware-video","slug":"hitea-hierarchical-temporal-aware-video","title":"HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training","date":"2022-12-30","arxiv_id":"2212.14546","repositories_listed":0,"syntology":null},{"url":"/paper/simvtp-simple-video-text-pre-training-with","slug":"simvtp-simple-video-text-pre-training-with","title":"SimVTP: Simple Video Text Pre-training with Masked Autoencoders","date":"2022-12-07","arxiv_id":"2212.03490","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-do-cross-view-and-cross-modal-alignment","title":"How do Cross-View and Cross-Modal Alignment Affect Representations in Contrastive Learning?","date":"2022-11-23","arxiv_id":"2211.13309","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaug-sparse-masked-autoencoder-for-efficient","title":"SMAUG: Sparse Masked Autoencoder for Efficient Video-Language Pre-training","date":"2022-11-21","arxiv_id":"2211.11446","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-hallucinating-vision-language-pre","title":"Learning by Hallucinating: Vision-Language Pre-training with Weak Supervision","date":"2022-10-24","arxiv_id":"2210.13591","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-semantic-alignment-network-for-1","title":"Fine-grained Semantic Alignment Network for Weakly Supervised Temporal Language Grounding","date":"2022-10-21","arxiv_id":"2210.11933","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-semantic-enhanced-interaction-for","title":"Cross-modal Semantic Enhanced Interaction for Image-Sentence Retrieval","date":"2022-10-17","arxiv_id":"2210.08908","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-referring-expression-comprehension-via","title":"Video Referring Expression Comprehension via Transformer with Content-aware Query","date":"2022-10-06","arxiv_id":"2210.02953","repositories_listed":0,"syntology":null},{"url":null,"slug":"jpg-jointly-learn-to-align-automated-disease","title":"JPG - Jointly Learn to Align: Automated Disease Prediction and Radiology Report Generation","date":"2022-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenflow-rethinking-fine-grained-cross-modal","title":"TokenFlow: Rethinking Fine-grained Cross-modal Alignment in Vision-Language Retrieval","date":"2022-09-28","arxiv_id":"2209.13822","repositories_listed":0,"syntology":null},{"url":"/paper/translation-scale-and-rotation-cross-modal","slug":"translation-scale-and-rotation-cross-modal","title":"Translation, Scale and Rotation: Cross-Modal Alignment Meets RGB-Infrared Vehicle Detection","date":"2022-09-28","arxiv_id":"2209.13801","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-cross-domain-alignment-network","title":"Multi-Modal Cross-Domain Alignment Network for Video Moment Retrieval","date":"2022-09-23","arxiv_id":"2209.11572","repositories_listed":0,"syntology":null},{"url":"/paper/omnivl-one-foundation-model-for-image","slug":"omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","arxiv_id":"2209.07526","repositories_listed":0,"syntology":null},{"url":null,"slug":"see-what-you-see-self-supervised-cross-modal","title":"See What You See: Self-supervised Cross-modal Retrieval of Visual Stimuli from Brain Activity","date":"2022-08-07","arxiv_id":"2208.03666","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-vision-and-language-modeling-for-multi","title":"Masked Vision and Language Modeling for Multi-modal Representation Learning","date":"2022-08-03","arxiv_id":"2208.02131","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-alignment-learning-of-vision","title":"Cross-Modal Alignment Learning of Vision-Language Conceptual Systems","date":"2022-07-31","arxiv_id":"2208.01744","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmixer-unpaired-vision-language-pre-training","title":"VLMixer: Unpaired Vision-Language Pre-training via Cross-Modal CutMix","date":"2022-06-17","arxiv_id":"2206.08919","repositories_listed":0,"syntology":null},{"url":null,"slug":"mslam-massively-multilingual-joint-pre","title":"mSLAM: Massively multilingual joint pre-training for speech and text","date":"2022-02-03","arxiv_id":"2202.01374","repositories_listed":0,"syntology":null},{"url":null,"slug":"kd-vlp-improving-end-to-end-vision-and-1","title":"KD-VLP: Improving End-to-End Vision-and-Language Pretraining with Object Knowledge Distillation","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-better-visual-representations-for","title":"Learning Better Visual Representations for Weakly-Supervised Object Detection Using Natural Language Supervision","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-joint-embedding-with-modality","title":"Learning Joint Embedding with Modality Alignments for Cross-Modal Retrieval of Recipes and Food Images","date":"2021-08-09","arxiv_id":"2108.03788","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-multi-modal-feature-embedding-and","title":"Structured Multi-modal Feature Embedding and Alignment for Image-Sentence Retrieval","date":"2021-08-05","arxiv_id":"2108.02417","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-in-cross-modal-retrieval","title":"Continual learning in cross-modal retrieval","date":"2021-04-14","arxiv_id":"2104.06806","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-intuitive-agent-for-remote-embodied","title":"Scene-Intuitive Agent for Remote Embodied Visual Grounding","date":"2021-03-24","arxiv_id":"2103.12944","repositories_listed":0,"syntology":null},{"url":null,"slug":"st-bert-cross-modal-language-model-pre","title":"ST-BERT: Cross-modal Language Model Pre-training For End-to-end Spoken Language Understanding","date":"2020-10-23","arxiv_id":"2010.12283","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-weakly-supervised","title":"Reinforcement Learning for Weakly Supervised Temporal Grounding of Natural Language in Untrimmed Videos","date":"2020-09-18","arxiv_id":"2009.08614","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-alignment-with-mixture-experts","title":"Cross-Modal Alignment with Mixture Experts Neural Network for Intral-City Retail Recommendation","date":"2020-09-17","arxiv_id":"2009.09926","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-modal-nonlinear-embeddings","title":"Learning Multi-Modal Nonlinear Embeddings: Performance Bounds and an Algorithm","date":"2020-06-03","arxiv_id":"2006.02330","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-cross-domain-moment-alignment","title":"Cross-Modal Cross-Domain Moment Alignment Network for Person Search","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"behind-the-scene-revealing-the-secrets-of-pre","title":"Behind the Scene: Revealing the Secrets of Pre-trained Vision-and-Language Models","date":"2020-05-15","arxiv_id":"2005.07310","repositories_listed":0,"syntology":null},{"url":"/paper/continuous-sign-language-recognition-through","slug":"continuous-sign-language-recognition-through","title":"Continuous Sign Language Recognition Through Cross-Modal Alignment of Video and Text Embeddings in a Joint-Latent Space","date":"2020-05-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mcqa-multimodal-co-attention-based-network","title":"MCQA: Multimodal Co-attention Based Network for Question Answering","date":"2020-04-25","arxiv_id":"2004.12238","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-audiovisual-learning","title":"Curriculum Audiovisual Learning","date":"2020-01-26","arxiv_id":"2001.09414","repositories_listed":0,"syntology":null},{"url":null,"slug":"acmm-aligned-cross-modal-memory-for-few-shot","title":"ACMM: Aligned Cross-Modal Memory for Few-Shot Image and Sentence Matching","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-and-match-networks-multi-domain-alignment","title":"Mix and match networks: cross-modal alignment for zero-pair image-to-image translation","date":"2019-03-08","arxiv_id":"1903.04294","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-cross-modal-alignment-of-speech","title":"Unsupervised Cross-Modal Alignment of Speech and Text Embedding Spaces","date":"2018-05-18","arxiv_id":"1805.07467","repositories_listed":0,"syntology":null}],"record_sha256":"2e83a35e722810fcf78b5fc1da5c889847ed2211438063b9d8e66a598543ae3f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}