{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/text-retrieval/papers/ran/1","list_of":"/task/text-retrieval","task":"Text Retrieval","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":117,"counts":{"archive_papers_tagged":671,"with_a_code_link":335,"where_syntology_ran_a_sample":117,"not_listed_spam_title":0,"listed":671,"listed_where_code_ran":117,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":99,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":99,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/text-retrieval/papers/ran/1","prev":null,"next":"/task/text-retrieval/papers/ran/2","papers":[{"url":"/paper/mstar-box-free-multi-query-scene-text","slug":"mstar-box-free-multi-query-scene-text","title":"MSTAR: Box-free Multi-query Scene Text Retrieval with Attention Recycling","date":"2025-06-12","arxiv_id":"2506.10609","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mstar-box-free-multi-query-scene-text#ran","syntology_url":"https://syntology.ai/paper/2506.10609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10609"}},"official":{"repos":["yingift/mstar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glap-general-contrastive-audio-text","slug":"glap-general-contrastive-audio-text","title":"GLAP: General contrastive audio-text pretraining across domains and languages","date":"2025-06-12","arxiv_id":"2506.11350","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/glap-general-contrastive-audio-text#ran","syntology_url":"https://syntology.ai/paper/2506.11350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.11350"}},"official":{"repos":["xiaomi-research/dasheng-glap"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/discovla-discrepancy-reduction-in-vision-1","slug":"discovla-discrepancy-reduction-in-vision-1","title":"DiscoVLA: Discrepancy Reduction in Vision, Language, and Alignment for Parameter-Efficient Video-Text Retrieval","date":"2025-06-10","arxiv_id":"2506.08887","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovla-discrepancy-reduction-in-vision-1#ran","syntology_url":"https://syntology.ai/paper/2506.08887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08887"}},"official":{"repos":["lunarshen/dsicovla"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fg-clip-fine-grained-visual-and-textual","slug":"fg-clip-fine-grained-visual-and-textual","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","date":"2025-05-08","arxiv_id":"2505.05071","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fg-clip-fine-grained-visual-and-textual#ran","syntology_url":"https://syntology.ai/paper/2505.05071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05071"}},"official":{"repos":["360cvgroup/fg-clip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/med3dvlm-an-efficient-vision-language-model","slug":"med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","arxiv_id":"2503.20047","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/med3dvlm-an-efficient-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.20047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20047"}},"official":{"repos":["mirthai/med3dvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/goal-global-local-object-alignment-learning","slug":"goal-global-local-object-alignment-learning","title":"GOAL: Global-local Object Alignment Learning","date":"2025-03-22","arxiv_id":"2503.17782","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/goal-global-local-object-alignment-learning#ran","syntology_url":"https://syntology.ai/paper/2503.17782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.17782"}},"official":{"repos":["perceptualai-lab/goal"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/peerqa-a-scientific-question-answering","slug":"peerqa-a-scientific-question-answering","title":"PeerQA: A Scientific Question Answering Dataset from Peer Reviews","date":"2025-02-19","arxiv_id":"2502.13668","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/peerqa-a-scientific-question-answering#ran","syntology_url":"https://syntology.ai/paper/2502.13668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13668"}},"official":{"repos":["ukplab/peerqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedica-an-open-biomedical-image-caption","slug":"biomedica-an-open-biomedical-image-caption","title":"BIOMEDICA: An Open Biomedical Image-Caption Archive, Dataset, and Vision-Language Models Derived from Scientific Literature","date":"2025-01-13","arxiv_id":"2501.07171","repositories_listed":2,"syntology":{"n":20,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/biomedica-an-open-biomedical-image-caption#ran","syntology_url":"https://syntology.ai/paper/2501.07171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07171"}},"official":{"repos":["ale9806/open_clip_with_biomedica","minwoosun/biomedica-etl"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gramian-multimodal-representation-learning","slug":"gramian-multimodal-representation-learning","title":"Gramian Multimodal Representation Learning and Alignment","date":"2024-12-16","arxiv_id":"2412.11959","repositories_listed":2,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gramian-multimodal-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2412.11959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11959"}},"official":{"repos":["ispamm/GRAM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/audiosetcaps-an-enriched-audio-caption","slug":"audiosetcaps-an-enriched-audio-caption","title":"AudioSetCaps: An Enriched Audio-Caption Dataset using Automated Generation Pipeline with Large Audio and Language Models","date":"2024-11-28","arxiv_id":"2411.18953","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiosetcaps-an-enriched-audio-caption#ran","syntology_url":"https://syntology.ai/paper/2411.18953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18953"}},"official":{"repos":["jishengbai/audiosetcaps"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/partial-scene-text-retrieval","slug":"partial-scene-text-retrieval","title":"Partial Scene Text Retrieval","date":"2024-11-15","arxiv_id":"2411.10261","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":15,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/partial-scene-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2411.10261","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.10261"}},"official":{"repos":["lanfeng4659/pstr"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-text-optimizing-rag-with-multimodal","slug":"beyond-text-optimizing-rag-with-multimodal","title":"Beyond Text: Optimizing RAG with Multimodal Inputs for Industrial Applications","date":"2024-10-29","arxiv_id":"2410.21943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-text-optimizing-rag-with-multimodal#ran","syntology_url":"https://syntology.ai/paper/2410.21943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21943"}},"official":{"repos":["riedlerm/multimodal_rag_for_industry"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/focus-distinguish-and-prompt-unleashing-clip","slug":"focus-distinguish-and-prompt-unleashing-clip","title":"Focus, Distinguish, and Prompt: Unleashing CLIP for Efficient and Flexible Scene Text Retrieval","date":"2024-08-01","arxiv_id":"2408.00441","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/focus-distinguish-and-prompt-unleashing-clip#ran","syntology_url":"https://syntology.ai/paper/2408.00441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00441"}},"official":{"repos":["gyann-z/fdp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mgte-generalized-long-context-text","slug":"mgte-generalized-long-context-text","title":"mGTE: Generalized Long-Context Text Representation and Reranking Models for Multilingual Text Retrieval","date":"2024-07-29","arxiv_id":"2407.19669","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mgte-generalized-long-context-text#ran","syntology_url":"https://syntology.ai/paper/2407.19669","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19669"}},"official":null}},{"url":"/paper/multi-label-cluster-discrimination-for-visual","slug":"multi-label-cluster-discrimination-for-visual","title":"Multi-label Cluster Discrimination for Visual Representation Learning","date":"2024-07-24","arxiv_id":"2407.17331","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":7,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-label-cluster-discrimination-for-visual#ran","syntology_url":"https://syntology.ai/paper/2407.17331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17331"}},"official":{"repos":["deepglint/unicom"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/object-aware-query-perturbation-for-cross","slug":"object-aware-query-perturbation-for-cross","title":"Object-Aware Query Perturbation for Cross-Modal Image-Text Retrieval","date":"2024-07-17","arxiv_id":"2407.12346","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-query-perturbation-for-cross#ran","syntology_url":"https://syntology.ai/paper/2407.12346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12346"}},"official":{"repos":["nec-n-sogi/query-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bm25s-orders-of-magnitude-faster-lexical","slug":"bm25s-orders-of-magnitude-faster-lexical","title":"BM25S: Orders of magnitude faster lexical search via eager sparse scoring","date":"2024-07-04","arxiv_id":"2407.03618","repositories_listed":3,"syntology":{"n":22,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/bm25s-orders-of-magnitude-faster-lexical#ran","syntology_url":"https://syntology.ai/paper/2407.03618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03618"}},"official":{"repos":["xhluca/bm25s"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/composing-object-relations-and-attributes-for-1","slug":"composing-object-relations-and-attributes-for-1","title":"Composing Object Relations and Attributes for Image-Text Matching","date":"2024-06-17","arxiv_id":"2406.11820","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/composing-object-relations-and-attributes-for-1#ran","syntology_url":"https://syntology.ai/paper/2406.11820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11820"}},"official":{"repos":["vkhoi/cora_cvpr24"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bivlc-extending-vision-language","slug":"bivlc-extending-vision-language","title":"BiVLC: Extending Vision-Language Compositionality Evaluation with Text-to-Image Retrieval","date":"2024-06-14","arxiv_id":"2406.09952","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bivlc-extending-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.09952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09952"}},"official":{"repos":["imirandam/bivlc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rwkv-clip-a-robust-vision-language","slug":"rwkv-clip-a-robust-vision-language","title":"RWKV-CLIP: A Robust Vision-Language Representation Learner","date":"2024-06-11","arxiv_id":"2406.06973","repositories_listed":2,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/rwkv-clip-a-robust-vision-language#ran","syntology_url":"https://syntology.ai/paper/2406.06973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06973"}},"official":{"repos":["deepglint/rwkv-clip"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/ldmol-text-conditioned-molecule-diffusion","slug":"ldmol-text-conditioned-molecule-diffusion","title":"LDMol: Text-to-Molecule Diffusion Model with Structurally Informative Latent Space","date":"2024-05-28","arxiv_id":"2405.17829","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":2,"n_ran_checked":5,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ldmol-text-conditioned-molecule-diffusion#ran","syntology_url":"https://syntology.ai/paper/2405.17829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17829"}},"official":{"repos":["jinhojsk515/ldmol"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/accelerating-transformers-with-spectrum-1","slug":"accelerating-transformers-with-spectrum-1","title":"Accelerating Transformers with Spectrum-Preserving Token Merging","date":"2024-05-25","arxiv_id":"2405.16148","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/accelerating-transformers-with-spectrum-1#ran","syntology_url":"https://syntology.ai/paper/2405.16148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16148"}},"official":{"repos":["hchautran/PiToMe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prott3-protein-to-text-generation-for-text","slug":"prott3-protein-to-text-generation-for-text","title":"ProtT3: Protein-to-Text Generation for Text-based Protein Understanding","date":"2024-05-21","arxiv_id":"2405.12564","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/prott3-protein-to-text-generation-for-text#ran","syntology_url":"https://syntology.ai/paper/2405.12564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12564"}},"official":{"repos":["acharkq/prott3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-deep-audio-text-retrieval-through","slug":"revisiting-deep-audio-text-retrieval-through","title":"Revisiting Deep Audio-Text Retrieval Through the Lens of Transportation","date":"2024-05-16","arxiv_id":"2405.10084","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/revisiting-deep-audio-text-retrieval-through#ran","syntology_url":"https://syntology.ai/paper/2405.10084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10084"}},"official":{"repos":["v-manhlt3/m-ltm-audio-text-retrieval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-text-similarity-in-transformer","slug":"explaining-text-similarity-in-transformer","title":"Explaining Text Similarity in Transformer Models","date":"2024-05-10","arxiv_id":"2405.06604","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explaining-text-similarity-in-transformer#ran","syntology_url":"https://syntology.ai/paper/2405.06604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06604"}},"official":{"repos":["alevas/xai_similarity_transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-inverted-indexes-for-approximate","slug":"efficient-inverted-indexes-for-approximate","title":"Efficient Inverted Indexes for Approximate Retrieval over Learned Sparse Representations","date":"2024-04-29","arxiv_id":"2404.18812","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-inverted-indexes-for-approximate#ran","syntology_url":"https://syntology.ai/paper/2404.18812","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18812"}},"official":{"repos":["tuskanny/seismic"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/efficient-remote-sensing-with-harmonized","slug":"efficient-remote-sensing-with-harmonized","title":"Efficient Remote Sensing with Harmonized Transfer Learning and Modality Alignment","date":"2024-04-28","arxiv_id":"2404.18253","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/efficient-remote-sensing-with-harmonized#ran","syntology_url":"https://syntology.ai/paper/2404.18253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18253"}},"official":{"repos":["seekerhuang/harma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/m3d-advancing-3d-medical-image-analysis-with","slug":"m3d-advancing-3d-medical-image-analysis-with","title":"M3D: Advancing 3D Medical Image Analysis with Multi-Modal Large Language Models","date":"2024-03-31","arxiv_id":"2404.00578","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3d-advancing-3d-medical-image-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2404.00578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00578"}},"official":{"repos":["baai-dcai/m3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic","slug":"arabicaqa-a-comprehensive-dataset-for-arabic","title":"ArabicaQA: A Comprehensive Dataset for Arabic Question Answering","date":"2024-03-26","arxiv_id":"2403.17848","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic#ran","syntology_url":"https://syntology.ai/paper/2403.17848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17848"}},"official":{"repos":["datascienceuibk/arabicaqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamlip-language-image-pre-training-with","slug":"dreamlip-language-image-pre-training-with","title":"DreamLIP: Language-Image Pre-training with Long Captions","date":"2024-03-25","arxiv_id":"2403.17007","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dreamlip-language-image-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2403.17007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17007"}},"official":{"repos":["zyf0619sjtu/DreamLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vid-tldr-training-free-token-merging-for","slug":"vid-tldr-training-free-token-merging-for","title":"vid-TLDR: Training Free Token merging for Light-weight Video Transformer","date":"2024-03-20","arxiv_id":"2403.13347","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vid-tldr-training-free-token-merging-for#ran","syntology_url":"https://syntology.ai/paper/2403.13347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13347"}},"official":{"repos":["mlvlab/vid-tldr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eye-gaze-guided-multi-modal-alignment","slug":"eye-gaze-guided-multi-modal-alignment","title":"Eye-gaze Guided Multi-modal Alignment for Medical Representation Learning","date":"2024-03-19","arxiv_id":"2403.12416","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/eye-gaze-guided-multi-modal-alignment#ran","syntology_url":"https://syntology.ai/paper/2403.12416","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12416"}},"official":{"repos":["momarky/egma"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multimodal-learned-sparse-retrieval-with","slug":"multimodal-learned-sparse-retrieval-with","title":"Multimodal Learned Sparse Retrieval with Probabilistic Expansion Control","date":"2024-02-27","arxiv_id":"2402.17535","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-learned-sparse-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2402.17535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17535"}},"official":{"repos":["thongnt99/lsr-multimodal"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embracing-language-inclusivity-and-diversity","slug":"embracing-language-inclusivity-and-diversity","title":"Embracing Language Inclusivity and Diversity in CLIP through Continual Language Learning","date":"2024-01-30","arxiv_id":"2401.17186","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embracing-language-inclusivity-and-diversity#ran","syntology_url":"https://syntology.ai/paper/2401.17186","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17186"}},"official":{"repos":["yangbang18/clfm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-3d-molecule-text-interpretation-in","slug":"towards-3d-molecule-text-interpretation-in","title":"Towards 3D Molecule-Text Interpretation in Language Models","date":"2024-01-25","arxiv_id":"2401.13923","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-3d-molecule-text-interpretation-in#ran","syntology_url":"https://syntology.ai/paper/2401.13923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13923"}},"official":{"repos":["lsh0520/3d-molm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-the-impact-of-false-negatives-in","slug":"mitigating-the-impact-of-false-negatives-in","title":"Mitigating the Impact of False Negatives in Dense Retrieval with Contrastive Confidence Regularization","date":"2023-12-30","arxiv_id":"2401.00165","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-the-impact-of-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2401.00165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00165"}},"official":{"repos":["wangskygit/passage-sieve"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/internvl-scaling-up-vision-foundation-models","slug":"internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/internvl-scaling-up-vision-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2312.14238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14238"}},"official":{"repos":["opengvlab/internvl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/synthesize-diagnose-and-optimize-towards-fine","slug":"synthesize-diagnose-and-optimize-towards-fine","title":"Synthesize, Diagnose, and Optimize: Towards Fine-Grained Vision-Language Understanding","date":"2023-11-30","arxiv_id":"2312.00081","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/synthesize-diagnose-and-optimize-towards-fine#ran","syntology_url":"https://syntology.ai/paper/2312.00081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00081"}},"official":{"repos":["wjpoom/spec"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/glen-generative-retrieval-via-lexical-index","slug":"glen-generative-retrieval-via-lexical-index","title":"GLEN: Generative Retrieval via Lexical Index Learning","date":"2023-11-06","arxiv_id":"2311.03057","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/glen-generative-retrieval-via-lexical-index#ran","syntology_url":"https://syntology.ai/paper/2311.03057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03057"}},"official":{"repos":["skleee/GLEN"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/unlock-multi-modal-capability-of-dense","slug":"unlock-multi-modal-capability-of-dense","title":"MARVEL: Unlocking the Multi-Modal Capability of Dense Retrieval via Visual Module Plugin","date":"2023-10-21","arxiv_id":"2310.14037","repositories_listed":1,"syntology":{"n":18,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlock-multi-modal-capability-of-dense#ran","syntology_url":"https://syntology.ai/paper/2310.14037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14037"}},"official":{"repos":["openmatch/marvel"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/molca-molecular-graph-language-modeling-with","slug":"molca-molecular-graph-language-modeling-with","title":"MolCA: Molecular Graph-Language Modeling with Cross-Modal Projector and Uni-Modal Adapter","date":"2023-10-19","arxiv_id":"2310.12798","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":16,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/molca-molecular-graph-language-modeling-with#ran","syntology_url":"https://syntology.ai/paper/2310.12798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12798"}},"official":{"repos":["acharkq/molca"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/frozen-transformers-in-language-models-are","slug":"frozen-transformers-in-language-models-are","title":"Frozen Transformers in Language Models Are Effective Visual Encoder Layers","date":"2023-10-19","arxiv_id":"2310.12973","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/frozen-transformers-in-language-models-are#ran","syntology_url":"https://syntology.ai/paper/2310.12973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12973"}},"official":{"repos":["ziqipang/lm4visualencoding"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/extending-multi-modal-contrastive","slug":"extending-multi-modal-contrastive","title":"Extending Multi-modal Contrastive Representations","date":"2023-10-13","arxiv_id":"2310.08884","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extending-multi-modal-contrastive#ran","syntology_url":"https://syntology.ai/paper/2310.08884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08884"}},"official":{"repos":["mcr-peft/ex-mcr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-llama-for-multi-stage-text","slug":"fine-tuning-llama-for-multi-stage-text","title":"Fine-Tuning LLaMA for Multi-Stage Text Retrieval","date":"2023-10-12","arxiv_id":"2310.08319","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fine-tuning-llama-for-multi-stage-text#ran","syntology_url":"https://syntology.ai/paper/2310.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08319"}},"official":null}},{"url":"/paper/languagebind-extending-video-language","slug":"languagebind-extending-video-language","title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","date":"2023-10-03","arxiv_id":"2310.01852","repositories_listed":6,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/languagebind-extending-video-language#ran","syntology_url":"https://syntology.ai/paper/2310.01852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01852"}},"official":{"repos":["pku-yuangroup/languagebind"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/prototype-based-aleatoric-uncertainty-1","slug":"prototype-based-aleatoric-uncertainty-1","title":"Prototype-based Aleatoric Uncertainty Quantification for Cross-modal Retrieval","date":"2023-09-29","arxiv_id":"2309.17093","repositories_listed":1,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":7,"n_pointer_only":8,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 1 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prototype-based-aleatoric-uncertainty-1#ran","syntology_url":"https://syntology.ai/paper/2309.17093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17093"}},"official":{"repos":["leolee99/pau"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unified-coarse-to-fine-alignment-for-video","slug":"unified-coarse-to-fine-alignment-for-video","title":"Unified Coarse-to-Fine Alignment for Video-Text Retrieval","date":"2023-09-18","arxiv_id":"2309.10091","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unified-coarse-to-fine-alignment-for-video#ran","syntology_url":"https://syntology.ai/paper/2309.10091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10091"}},"official":{"repos":["ziyang412/ucofia"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-event-video-text-retrieval","slug":"multi-event-video-text-retrieval","title":"Multi-event Video-Text Retrieval","date":"2023-08-22","arxiv_id":"2308.11551","repositories_listed":1,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":9,"n_instrument":6,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-event-video-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2308.11551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11551"}},"official":{"repos":["gengyuanmax/mevtr"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/alip-adaptive-language-image-pre-training","slug":"alip-adaptive-language-image-pre-training","title":"ALIP: Adaptive Language-Image Pre-training with Synthetic Caption","date":"2023-08-16","arxiv_id":"2308.08428","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/alip-adaptive-language-image-pre-training#ran","syntology_url":"https://syntology.ai/paper/2308.08428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08428"}},"official":{"repos":["deepglint/alip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-dataset-distillation-for-image","slug":"multimodal-dataset-distillation-for-image","title":"Vision-Language Dataset Distillation","date":"2023-08-15","arxiv_id":"2308.07545","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-dataset-distillation-for-image#ran","syntology_url":"https://syntology.ai/paper/2308.07545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07545"}},"official":{"repos":["princetonvisualai/multimodal_dataset_distillation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/helping-hands-an-object-aware-ego-centric","slug":"helping-hands-an-object-aware-ego-centric","title":"Helping Hands: An Object-Aware Ego-Centric Video Recognition Model","date":"2023-08-15","arxiv_id":"2308.07918","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/helping-hands-an-object-aware-ego-centric#ran","syntology_url":"https://syntology.ai/paper/2308.07918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07918"}},"official":{"repos":["chuhanxx/helping_hand_for_egocentric_videos"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/advclip-downstream-agnostic-adversarial","slug":"advclip-downstream-agnostic-adversarial","title":"AdvCLIP: Downstream-agnostic Adversarial Examples in Multimodal Contrastive Learning","date":"2023-08-14","arxiv_id":"2308.07026","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":10,"n_pointer_only":5,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 1 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/advclip-downstream-agnostic-adversarial#ran","syntology_url":"https://syntology.ai/paper/2308.07026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07026"}},"official":{"repos":["cgcl-codes/advclip"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/set-level-guidance-attack-boosting","slug":"set-level-guidance-attack-boosting","title":"Set-level Guidance Attack: Boosting Adversarial Transferability of Vision-Language Pre-training Models","date":"2023-07-26","arxiv_id":"2307.14061","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":1,"n_ran_checked":4,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/set-level-guidance-attack-boosting#ran","syntology_url":"https://syntology.ai/paper/2307.14061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14061"}},"official":{"repos":["Zoky-2020/Set-level_Guidance_Attack"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/prior-prototype-representation-joint-learning","slug":"prior-prototype-representation-joint-learning","title":"PRIOR: Prototype Representation Joint Learning from Medical Images and Reports","date":"2023-07-24","arxiv_id":"2307.12577","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prior-prototype-representation-joint-learning#ran","syntology_url":"https://syntology.ai/paper/2307.12577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12577"}},"official":{"repos":["qtacierp/prior"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-pre-training-adapt-visual-language","slug":"stop-pre-training-adapt-visual-language","title":"Stop Pre-Training: Adapt Visual-Language Models to Unseen Languages","date":"2023-06-29","arxiv_id":"2306.16774","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-pre-training-adapt-visual-language#ran","syntology_url":"https://syntology.ai/paper/2306.16774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16774"}},"official":{"repos":["yasminekaroui/clicotea"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs5m-a-large-scale-vision-language-dataset","slug":"rs5m-a-large-scale-vision-language-dataset","title":"RS5M and GeoRSCLIP: A Large Scale Vision-Language Dataset and A Large Vision-Language Model for Remote Sensing","date":"2023-06-20","arxiv_id":"2306.11300","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs5m-a-large-scale-vision-language-dataset#ran","syntology_url":"https://syntology.ai/paper/2306.11300","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11300"}},"official":{"repos":["om-ai-lab/rs5m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-token-guided-image-text-retrieval","slug":"efficient-token-guided-image-text-retrieval","title":"Efficient Token-Guided Image-Text Retrieval with Consistent Multimodal Contrastive Training","date":"2023-06-15","arxiv_id":"2306.08789","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/efficient-token-guided-image-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2306.08789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08789"}},"official":null}},{"url":"/paper/contrasting-intra-modal-and-ranking-cross","slug":"contrasting-intra-modal-and-ranking-cross","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","date":"2023-06-15","arxiv_id":"2306.08832","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contrasting-intra-modal-and-ranking-cross#ran","syntology_url":"https://syntology.ai/paper/2306.08832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08832"}},"official":{"repos":["lezhang7/Enhance-FineGrained","magiccircuit/enhance-finegrained"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/babel-imagenet-massively-multilingual","slug":"babel-imagenet-massively-multilingual","title":"Babel-ImageNet: Massively Multilingual Evaluation of Vision-and-Language Representations","date":"2023-06-14","arxiv_id":"2306.08658","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/babel-imagenet-massively-multilingual#ran","syntology_url":"https://syntology.ai/paper/2306.08658","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08658"}},"official":{"repos":["gregor-ge/babel-imagenet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/test-time-adaptation-with-clip-reward-for","slug":"test-time-adaptation-with-clip-reward-for","title":"Test-Time Adaptation with CLIP Reward for Zero-Shot Generalization in Vision-Language Models","date":"2023-05-29","arxiv_id":"2305.18010","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/test-time-adaptation-with-clip-reward-for#ran","syntology_url":"https://syntology.ai/paper/2305.18010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18010"}},"official":{"repos":["mzhaoshuai/rlcf"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/s-clip-semi-supervised-vision-language-1","slug":"s-clip-semi-supervised-vision-language-1","title":"S-CLIP: Semi-supervised Vision-Language Learning using Few Specialist Captions","date":"2023-05-23","arxiv_id":"2305.14095","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/s-clip-semi-supervised-vision-language-1#ran","syntology_url":"https://syntology.ai/paper/2305.14095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14095"}},"official":{"repos":["alinlab/s-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/one-peace-exploring-one-general","slug":"one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","arxiv_id":"2305.11172","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/one-peace-exploring-one-general#ran","syntology_url":"https://syntology.ai/paper/2305.11172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11172"}},"official":{"repos":["OFA-Sys/ONE-PEACE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/region-aware-pretraining-for-open-vocabulary","slug":"region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","arxiv_id":"2305.07011","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/region-aware-pretraining-for-open-vocabulary#ran","syntology_url":"https://syntology.ai/paper/2305.07011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07011"}},"official":{"repos":["mcahny/rovit","google-research/google-research"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperbolic-image-text-representations","slug":"hyperbolic-image-text-representations","title":"Hyperbolic Image-Text Representations","date":"2023-04-18","arxiv_id":"2304.09172","repositories_listed":2,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hyperbolic-image-text-representations#ran","syntology_url":"https://syntology.ai/paper/2304.09172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09172"}},"official":{"repos":["facebookresearch/meru"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/equivariant-similarity-for-vision-language","slug":"equivariant-similarity-for-vision-language","title":"Equivariant Similarity for Vision-Language Foundation Models","date":"2023-03-25","arxiv_id":"2303.14465","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/equivariant-similarity-for-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.14465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14465"}},"official":{"repos":["wangt-cn/eqben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pmc-clip-contrastive-language-image-pre","slug":"pmc-clip-contrastive-language-image-pre","title":"PMC-CLIP: Contrastive Language-Image Pre-training using Biomedical Documents","date":"2023-03-13","arxiv_id":"2303.07240","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/pmc-clip-contrastive-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2303.07240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.07240"}},"official":{"repos":["WeixiongLin/PMC-CLIP"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/uniadapter-unified-parameter-efficient","slug":"uniadapter-unified-parameter-efficient","title":"UniAdapter: Unified Parameter-Efficient Transfer Learning for Cross-modal Modeling","date":"2023-02-13","arxiv_id":"2302.06605","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uniadapter-unified-parameter-efficient#ran","syntology_url":"https://syntology.ai/paper/2302.06605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06605"}},"official":{"repos":["rerv/uniadapter","uniadapter/uniadapter"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-molecule-structure-text-model-for","slug":"multi-modal-molecule-structure-text-model-for","title":"Multi-modal Molecule Structure-text Model for Text-based Retrieval and Editing","date":"2022-12-21","arxiv_id":"2212.10789","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-modal-molecule-structure-text-model-for#ran","syntology_url":"https://syntology.ai/paper/2212.10789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10789"}},"official":{"repos":["chao1224/moleculestm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/flexivit-one-model-for-all-patch-sizes","slug":"flexivit-one-model-for-all-patch-sizes","title":"FlexiViT: One Model for All Patch Sizes","date":"2022-12-15","arxiv_id":"2212.08013","repositories_listed":6,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flexivit-one-model-for-all-patch-sizes#ran","syntology_url":"https://syntology.ai/paper/2212.08013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08013"}},"official":{"repos":["google-research/big_vision"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/coco-dr-combating-distribution-shifts-in-zero","slug":"coco-dr-combating-distribution-shifts-in-zero","title":"COCO-DR: Combating Distribution Shifts in Zero-Shot Dense Retrieval with Contrastive and Distributionally Robust Learning","date":"2022-10-27","arxiv_id":"2210.15212","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/coco-dr-combating-distribution-shifts-in-zero#ran","syntology_url":"https://syntology.ai/paper/2210.15212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15212"}},"official":{"repos":["openmatch/coco-dr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/simans-simple-ambiguous-negatives-sampling","slug":"simans-simple-ambiguous-negatives-sampling","title":"SimANS: Simple Ambiguous Negatives Sampling for Dense Text Retrieval","date":"2022-10-21","arxiv_id":"2210.11773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simans-simple-ambiguous-negatives-sampling#ran","syntology_url":"https://syntology.ai/paper/2210.11773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11773"}},"official":{"repos":["microsoft/simxns"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtc-improving-video-text-retrieval-with-user","slug":"vtc-improving-video-text-retrieval-with-user","title":"VTC: Improving Video-Text Retrieval with User Comments","date":"2022-10-19","arxiv_id":"2210.10820","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtc-improving-video-text-retrieval-with-user#ran","syntology_url":"https://syntology.ai/paper/2210.10820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.10820"}},"official":null}},{"url":"/paper/mteb-massive-text-embedding-benchmark","slug":"mteb-massive-text-embedding-benchmark","title":"MTEB: Massive Text Embedding Benchmark","date":"2022-10-13","arxiv_id":"2210.07316","repositories_listed":5,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mteb-massive-text-embedding-benchmark#ran","syntology_url":"https://syntology.ai/paper/2210.07316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07316"}},"official":{"repos":["embeddings-benchmark/mteb"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mixed-modality-representation-learning-and","slug":"mixed-modality-representation-learning-and","title":"Mixed-modality Representation Learning and Pre-training for Joint Table-and-Text Retrieval in OpenQA","date":"2022-10-11","arxiv_id":"2210.05197","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mixed-modality-representation-learning-and#ran","syntology_url":"https://syntology.ai/paper/2210.05197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.05197"}},"official":{"repos":["jun-jie-huang/otter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/clip-vip-adapting-pre-trained-image-text","slug":"clip-vip-adapting-pre-trained-image-text","title":"CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment","date":"2022-09-14","arxiv_id":"2209.06430","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clip-vip-adapting-pre-trained-image-text#ran","syntology_url":"https://syntology.ai/paper/2209.06430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06430"}},"official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-multi-modality-retrieval-with-one","slug":"universal-multi-modality-retrieval-with-one","title":"Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval","date":"2022-09-01","arxiv_id":"2209.00179","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/universal-multi-modality-retrieval-with-one#ran","syntology_url":"https://syntology.ai/paper/2209.00179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00179"}},"official":{"repos":["openmatch/univl-dr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/x-clip-end-to-end-multi-grained-contrastive","slug":"x-clip-end-to-end-multi-grained-contrastive","title":"X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval","date":"2022-07-15","arxiv_id":"2207.07285","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/x-clip-end-to-end-multi-grained-contrastive#ran","syntology_url":"https://syntology.ai/paper/2207.07285","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07285"}},"official":{"repos":["xuguohai/X-CLIP"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/parameter-efficient-prompt-tuning-makes","slug":"parameter-efficient-prompt-tuning-makes","title":"Parameter-Efficient Prompt Tuning Makes Generalized and Calibrated Neural Text Retrievers","date":"2022-07-14","arxiv_id":"2207.07087","repositories_listed":2,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/parameter-efficient-prompt-tuning-makes#ran","syntology_url":"https://syntology.ai/paper/2207.07087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.07087"}},"official":{"repos":["thudm/p-tuning-v2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/egocentric-video-language-pretraining","slug":"egocentric-video-language-pretraining","title":"Egocentric Video-Language Pretraining","date":"2022-06-03","arxiv_id":"2206.01670","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egocentric-video-language-pretraining#ran","syntology_url":"https://syntology.ai/paper/2206.01670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01670"}},"official":{"repos":["showlab/egovlp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/cross-view-language-modeling-towards-unified","slug":"cross-view-language-modeling-towards-unified","title":"Cross-View Language Modeling: Towards Unified Cross-Lingual Cross-Modal Pre-training","date":"2022-06-01","arxiv_id":"2206.00621","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cross-view-language-modeling-towards-unified#ran","syntology_url":"https://syntology.ai/paper/2206.00621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00621"}},"official":{"repos":["zengyan-97/cclm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-and-r2d2-a-large-scale-chinese-cross","slug":"zero-and-r2d2-a-large-scale-chinese-cross","title":"CCMB: A Large-scale Chinese Cross-modal Benchmark","date":"2022-05-08","arxiv_id":"2205.03860","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zero-and-r2d2-a-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2205.03860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.03860"}},"official":{"repos":["yuxie11/R2D2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-contrastive-learning-for-speech-1","slug":"cross-modal-contrastive-learning-for-speech-1","title":"Cross-modal Contrastive Learning for Speech Translation","date":"2022-05-05","arxiv_id":"2205.02444","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-contrastive-learning-for-speech-1#ran","syntology_url":"https://syntology.ai/paper/2205.02444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.02444"}},"official":{"repos":["reneeye/const"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wukong-100-million-large-scale-chinese-cross","slug":"wukong-100-million-large-scale-chinese-cross","title":"Wukong: A 100 Million Large-scale Chinese Cross-modal Pre-training Benchmark","date":"2022-02-14","arxiv_id":"2202.06767","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wukong-100-million-large-scale-chinese-cross#ran","syntology_url":"https://syntology.ai/paper/2202.06767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.06767"}},"official":null}},{"url":"/paper/dall-eval-probing-the-reasoning-skills-and","slug":"dall-eval-probing-the-reasoning-skills-and","title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generation Models","date":"2022-02-08","arxiv_id":"2202.04053","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":5,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dall-eval-probing-the-reasoning-skills-and#ran","syntology_url":"https://syntology.ai/paper/2202.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04053"}},"official":{"repos":["j-min/dalleval"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","slug":"bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_constructed":8,"n_ran_checked":9,"n_instrument":4,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"13 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/bridgeformer-bridging-video-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2201.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.04850"}},"official":{"repos":["tencentarc/mcq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/clip-lite-information-efficient-visual","slug":"clip-lite-information-efficient-visual","title":"CLIP-Lite: Information Efficient Visual Representation Learning with Language Supervision","date":"2021-12-14","arxiv_id":"2112.07133","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip-lite-information-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2112.07133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07133"}},"official":{"repos":["4m4n5/CLIP-Lite"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-retriever-ranker-for-dense-text","slug":"adversarial-retriever-ranker-for-dense-text","title":"Adversarial Retriever-Ranker for dense text retrieval","date":"2021-10-07","arxiv_id":"2110.03611","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adversarial-retriever-ranker-for-dense-text#ran","syntology_url":"https://syntology.ai/paper/2110.03611","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.03611"}},"official":{"repos":["microsoft/ar2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-stage-pre-training-over-simplified","slug":"multi-stage-pre-training-over-simplified","title":"Multi-stage Pre-training over Simplified Multimodal Pre-training Models","date":"2021-07-22","arxiv_id":"2107.14596","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/multi-stage-pre-training-over-simplified#ran","syntology_url":"https://syntology.ai/paper/2107.14596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14596"}},"official":{"repos":["lttsmn/LXMERT-S"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/align-before-fuse-vision-and-language","slug":"align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","arxiv_id":"2107.07651","repositories_listed":6,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/align-before-fuse-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2107.07651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07651"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/more-robust-dense-retrieval-with-contrastive","slug":"more-robust-dense-retrieval-with-contrastive","title":"More Robust Dense Retrieval with Contrastive Dual Learning","date":"2021-07-16","arxiv_id":"2107.07773","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/more-robust-dense-retrieval-with-contrastive#ran","syntology_url":"https://syntology.ai/paper/2107.07773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07773"}},"official":{"repos":["thunlp/DANCE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-relation-alignment-for-calibrated","slug":"learning-relation-alignment-for-calibrated","title":"Learning Relation Alignment for Calibrated Cross-modal Retrieval","date":"2021-05-28","arxiv_id":"2105.13868","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-relation-alignment-for-calibrated#ran","syntology_url":"https://syntology.ai/paper/2105.13868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13868"}},"official":{"repos":["lancopku/IAIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/clip4clip-an-empirical-study-of-clip-for-end","slug":"clip4clip-an-empirical-study-of-clip-for-end","title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval","date":"2021-04-18","arxiv_id":"2104.08860","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clip4clip-an-empirical-study-of-clip-for-end#ran","syntology_url":"https://syntology.ai/paper/2104.08860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08860"}},"official":{"repos":["ArrowLuo/CLIP4Clip"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/beir-a-heterogenous-benchmark-for-zero-shot","slug":"beir-a-heterogenous-benchmark-for-zero-shot","title":"BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models","date":"2021-04-17","arxiv_id":"2104.08663","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beir-a-heterogenous-benchmark-for-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2104.08663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08663"}},"official":{"repos":["UKPLab/beir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"418f755dab9f0aa06d398b0920930cfd0bd8a705b389ea929805f0d359f7f93e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}