{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/coco/papers/2","list_of":"/dataset/coco","dataset":"COCO (Common Objects in Context)","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"archive","order_definition":"date (newest first), then slug","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":579,"counts":{"papers_with_a_benchmark_row":579,"with_a_code_link":504,"where_syntology_ran_a_sample":256,"not_listed_spam_title":0,"listed":579,"listed_where_code_ran":256,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":226,"every_run_a_failure_of_syntologys_instrument":30,"listed_with_a_run_with_no_instrument_failure":226,"listed_every_run_a_failure_of_syntologys_instrument":30,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/coco","prev":"/dataset/coco","next":"/dataset/coco/papers/3","papers":[{"paper":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}}}},{"paper":"/paper/rtmdet-an-empirical-study-of-designing-real","slug":"rtmdet-an-empirical-study-of-designing-real","title":"RTMDet: An Empirical Study of Designing Real-Time Object Detectors","date":"2022-12-14","arxiv_id":"2212.07784","rows_on_this_dataset":2,"code_links":14,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":16,"samples_constructed":0,"samples_ran_checked":16,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":3,"official":{"repos":["open-mmlab/mmdetection"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rtmdet-an-empirical-study-of-designing-real#ran","syntology_url":"https://syntology.ai/paper/2212.07784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07784"}}}},{"paper":"/paper/deepcut-unsupervised-segmentation-using-graph","slug":"deepcut-unsupervised-segmentation-using-graph","title":"DeepCut: Unsupervised Segmentation using Graph Neural Networks Clustering","date":"2022-12-12","arxiv_id":"2212.05853","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deepcut-unsupervised-segmentation-using-graph#ran","syntology_url":"https://syntology.ai/paper/2212.05853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05853"}}}},{"paper":"/paper/nms-strikes-back","slug":"nms-strikes-back","title":"NMS Strikes Back","date":"2022-12-12","arxiv_id":"2212.06137","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/resolving-semantic-confusions-for-improved-1","slug":"resolving-semantic-confusions-for-improved-1","title":"Resolving Semantic Confusions for Improved Zero-Shot Detection","date":"2022-12-12","arxiv_id":"2212.06097","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip","slug":"x-paste-revisit-copy-paste-at-scale-with-clip","title":"X-Paste: Revisiting Scalable Copy-Paste for Instance Segmentation using CLIP and StableDiffusion","date":"2022-12-07","arxiv_id":"2212.03863","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["yoctta/xpaste"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip#ran","syntology_url":"https://syntology.ai/paper/2212.03863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03863"}}}},{"paper":"/paper/diffusioninst-diffusion-model-for-instance","slug":"diffusioninst-diffusion-model-for-instance","title":"DiffusionInst: Diffusion Model for Instance Segmentation","date":"2022-12-06","arxiv_id":"2212.02773","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":16,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":5,"samples_unverified":4,"pointer_only_for_licence":7,"official":{"repos":["chenhaoxing/DiffusionInst"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/diffusioninst-diffusion-model-for-instance#ran","syntology_url":"https://syntology.ai/paper/2212.02773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02773"}}}},{"paper":"/paper/box2mask-box-supervised-instance-segmentation","slug":"box2mask-box-supervised-instance-segmentation","title":"Box2Mask: Box-supervised Instance Segmentation via Level-set Evolution","date":"2022-12-03","arxiv_id":"2212.01579","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/grit-a-generative-region-to-text-transformer","slug":"grit-a-generative-region-to-text-transformer","title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","date":"2022-12-01","arxiv_id":"2212.00280","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":8,"official":{"repos":["JialianW/GRiT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/grit-a-generative-region-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2212.00280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00280"}}}},{"paper":"/paper/segclip-patch-aggregation-with-learnable","slug":"segclip-patch-aggregation-with-learnable","title":"SegCLIP: Patch Aggregation with Learnable Centers for Open-Vocabulary Semantic Segmentation","date":"2022-11-27","arxiv_id":"2211.14813","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/rethinking-alignment-and-uniformity-in","slug":"rethinking-alignment-and-uniformity-in","title":"Rethinking Alignment and Uniformity in Unsupervised Semantic Segmentation","date":"2022-11-26","arxiv_id":"2211.14513","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/shifted-diffusion-for-text-to-image","slug":"shifted-diffusion-for-text-to-image","title":"Shifted Diffusion for Text-to-image Generation","date":"2022-11-24","arxiv_id":"2211.15388","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":3,"samples_unverified":4,"pointer_only_for_licence":6,"official":{"repos":["drboog/Shifted_Diffusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/shifted-diffusion-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2211.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15388"}}}},{"paper":"/paper/damo-yolo-a-report-on-real-time-object","slug":"damo-yolo-a-report-on-real-time-object","title":"DAMO-YOLO : A Report on Real-Time Object Detection Design","date":"2022-11-23","arxiv_id":"2211.15444","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":7,"pointer_only_for_licence":12,"official":{"repos":["alibaba/lightweight-neural-architecture-search","tinyvision/damo-yolo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/damo-yolo-a-report-on-real-time-object#ran","syntology_url":"https://syntology.ai/paper/2211.15444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15444"}}}},{"paper":"/paper/detrs-with-collaborative-hybrid-assignments","slug":"detrs-with-collaborative-hybrid-assignments","title":"DETRs with Collaborative Hybrid Assignments Training","date":"2022-11-22","arxiv_id":"2211.12860","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/detrs-with-collaborative-hybrid-assignments#ran","syntology_url":"https://syntology.ai/paper/2211.12860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12860"}}}},{"paper":"/paper/retrieval-augmented-multimodal-language","slug":"retrieval-augmented-multimodal-language","title":"Retrieval-Augmented Multimodal Language Modeling","date":"2022-11-22","arxiv_id":"2211.12561","rows_on_this_dataset":13,"code_links":0,"syntology":null},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":6,"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}}}},{"paper":"/paper/towards-all-in-one-pre-training-via","slug":"towards-all-in-one-pre-training-via","title":"Towards All-in-one Pre-training via Maximizing Multi-modal Mutual Information","date":"2022-11-17","arxiv_id":"2211.09807","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","slug":"eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","arxiv_id":"2211.07636","rows_on_this_dataset":4,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["baaivision/eva","rwightman/pytorch-image-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/eva-exploring-the-limits-of-masked-visual#ran","syntology_url":"https://syntology.ai/paper/2211.07636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07636"}}}},{"paper":"/paper/internimage-exploring-large-scale-vision","slug":"internimage-exploring-large-scale-vision","title":"InternImage: Exploring Large-Scale Vision Foundation Models with Deformable Convolutions","date":"2022-11-10","arxiv_id":"2211.05778","rows_on_this_dataset":11,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["opengvlab/internimage"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/internimage-exploring-large-scale-vision#ran","syntology_url":"https://syntology.ai/paper/2211.05778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.05778"}}}},{"paper":"/paper/oneformer-one-transformer-to-rule-universal","slug":"oneformer-one-transformer-to-rule-universal","title":"OneFormer: One Transformer to Rule Universal Image Segmentation","date":"2022-11-10","arxiv_id":"2211.06220","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["SHI-Labs/OneFormer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/oneformer-one-transformer-to-rule-universal#ran","syntology_url":"https://syntology.ai/paper/2211.06220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.06220"}}}},{"paper":"/paper/efficient-multi-order-gated-aggregation","slug":"efficient-multi-order-gated-aggregation","title":"MogaNet: Multi-order Gated Aggregation Network","date":"2022-11-07","arxiv_id":"2211.03295","rows_on_this_dataset":10,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":12,"samples_constructed":7,"samples_ran_checked":10,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["Westlake-AI/MogaNet","Westlake-AI/openmixup","chengtan9907/OpenSTL"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficient-multi-order-gated-aggregation#ran","syntology_url":"https://syntology.ai/paper/2211.03295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03295"}}}},{"paper":"/paper/group-detr-v2-strong-object-detector-with-1","slug":"group-detr-v2-strong-object-detector-with-1","title":"Group DETR v2: Strong Object Detector with Encoder-Decoder Pretraining","date":"2022-11-07","arxiv_id":"2211.03594","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/could-giant-pretrained-image-models-extract","slug":"could-giant-pretrained-image-models-extract","title":"Could Giant Pretrained Image Models Extract Universal Representations?","date":"2022-11-03","arxiv_id":"2211.02043","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/ediffi-text-to-image-diffusion-models-with-an","slug":"ediffi-text-to-image-diffusion-models-with-an","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","date":"2022-11-02","arxiv_id":"2211.01324","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ediffi-text-to-image-diffusion-models-with-an#ran","syntology_url":"https://syntology.ai/paper/2211.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.01324"}}}},{"paper":"/paper/ernie-vilg-2-0-improving-text-to-image","slug":"ernie-vilg-2-0-improving-text-to-image","title":"ERNIE-ViLG 2.0: Improving Text-to-Image Diffusion Model with Knowledge-Enhanced Mixture-of-Denoising-Experts","date":"2022-10-27","arxiv_id":"2210.15257","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/dissecting-deep-metric-learning-losses-for","slug":"dissecting-deep-metric-learning-losses-for","title":"Dissecting Deep Metric Learning Losses for Image-Text Retrieval","date":"2022-10-21","arxiv_id":"2210.13188","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/unsupervised-image-semantic-segmentation","slug":"unsupervised-image-semantic-segmentation","title":"Unsupervised Image Semantic Segmentation through Superpixels and Graph Neural Networks","date":"2022-10-21","arxiv_id":"2210.11810","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/towards-sustainable-self-supervised-learning","slug":"towards-sustainable-self-supervised-learning","title":"Towards Sustainable Self-supervised Learning","date":"2022-10-20","arxiv_id":"2210.11016","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":6,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":8,"official":{"repos":["sail-sg/tec"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/towards-sustainable-self-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11016"}}}},{"paper":"/paper/a-tri-layer-plugin-to-improve-occluded","slug":"a-tri-layer-plugin-to-improve-occluded","title":"A Tri-Layer Plugin to Improve Occluded Detection","date":"2022-10-18","arxiv_id":"2210.10046","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/perceptual-grouping-in-vision-language-models","slug":"perceptual-grouping-in-vision-language-models","title":"Perceptual Grouping in Contrastive Vision-Language Models","date":"2022-10-18","arxiv_id":"2210.09996","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/swinv2-imagen-hierarchical-vision-transformer","slug":"swinv2-imagen-hierarchical-vision-transformer","title":"Swinv2-Imagen: Hierarchical Vision Transformer Diffusion Models for Text-to-Image Generation","date":"2022-10-18","arxiv_id":"2210.09549","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/move-unsupervised-movable-object-segmentation","slug":"move-unsupervised-movable-object-segmentation","title":"MOVE: Unsupervised Movable Object Segmentation and Detection","date":"2022-10-14","arxiv_id":"2210.07920","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":1,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":6,"official":{"repos":["adambielski/move-seg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/move-unsupervised-movable-object-segmentation#ran","syntology_url":"https://syntology.ai/paper/2210.07920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07920"}}}},{"paper":"/paper/boxteacher-exploring-high-quality-pseudo","slug":"boxteacher-exploring-high-quality-pseudo","title":"BoxTeacher: Exploring High-Quality Pseudo Labels for Weakly Supervised Instance Segmentation","date":"2022-10-11","arxiv_id":"2210.05174","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/towards-discriminative-and-transferable-one","slug":"towards-discriminative-and-transferable-one","title":"Towards Discriminative and Transferable One-Stage Few-Shot Object Detectors","date":"2022-10-11","arxiv_id":"2210.05783","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/k-means-for-unsupervised-instance","slug":"k-means-for-unsupervised-instance","title":"K-means for unsupervised instance segmentation using a self-supervised transformer","date":"2022-10-04","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/moat-alternating-mobile-convolution-and","slug":"moat-alternating-mobile-convolution-and","title":"MOAT: Alternating Mobile Convolution and Attention Brings Strong Vision Models","date":"2022-10-04","arxiv_id":"2210.01820","rows_on_this_dataset":18,"code_links":2,"syntology":null},{"paper":"/paper/ernie-vil-2-0-multi-view-contrastive-learning","slug":"ernie-vil-2-0-multi-view-contrastive-learning","title":"ERNIE-ViL 2.0: Multi-view Contrastive Learning for Image-Text Pre-training","date":"2022-09-30","arxiv_id":"2209.15270","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/dilated-neighborhood-attention-transformer","slug":"dilated-neighborhood-attention-transformer","title":"Dilated Neighborhood Attention Transformer","date":"2022-09-29","arxiv_id":"2209.15001","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/re-imagen-retrieval-augmented-text-to-image","slug":"re-imagen-retrieval-augmented-text-to-image","title":"Re-Imagen: Retrieval-Augmented Text-to-Image Generator","date":"2022-09-29","arxiv_id":"2209.14491","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/all-are-worth-words-a-vit-backbone-for-score","slug":"all-are-worth-words-a-vit-backbone-for-score","title":"All are Worth Words: A ViT Backbone for Diffusion Models","date":"2022-09-25","arxiv_id":"2209.12152","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["baofff/U-ViT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/all-are-worth-words-a-vit-backbone-for-score#ran","syntology_url":"https://syntology.ai/paper/2209.12152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12152"}}}},{"paper":"/paper/omnivl-one-foundation-model-for-image","slug":"omnivl-one-foundation-model-for-image","title":"OmniVL:One Foundation Model for Image-Language and Video-Language Tasks","date":"2022-09-15","arxiv_id":"2209.07526","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/combining-metric-learning-and-attention-heads","slug":"combining-metric-learning-and-attention-heads","title":"Combining Metric Learning and Attention Heads For Accurate and Efficient Multilabel Image Classification","date":"2022-09-14","arxiv_id":"2209.06585","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/exploring-target-representations-for-masked","slug":"exploring-target-representations-for-masked","title":"Exploring Target Representations for Masked Autoencoders","date":"2022-09-08","arxiv_id":"2209.03917","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["liuxingbin/dbot"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/exploring-target-representations-for-masked#ran","syntology_url":"https://syntology.ai/paper/2209.03917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03917"}}}},{"paper":"/paper/yolov6-a-single-stage-object-detection","slug":"yolov6-a-single-stage-object-detection","title":"YOLOv6: A Single-Stage Object Detection Framework for Industrial Applications","date":"2022-09-07","arxiv_id":"2209.02976","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["meituan/yolov6"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov6-a-single-stage-object-detection#ran","syntology_url":"https://syntology.ai/paper/2209.02976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02976"}}}},{"paper":"/paper/dpit-dual-pipeline-integrated-transformer-for","slug":"dpit-dual-pipeline-integrated-transformer-for","title":"DPIT: Dual-Pipeline Integrated Transformer for Human Pose Estimation","date":"2022-09-02","arxiv_id":"2209.02431","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/gswin-gated-mlp-vision-model-with","slug":"gswin-gated-mlp-vision-model-with","title":"gSwin: Gated MLP Vision Model with Hierarchical Structure of Shifted Window","date":"2022-08-24","arxiv_id":"2208.11718","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/image-as-a-foreign-language-beit-pretraining","slug":"image-as-a-foreign-language-beit-pretraining","title":"Image as a Foreign Language: BEiT Pretraining for All Vision and Vision-Language Tasks","date":"2022-08-22","arxiv_id":"2208.10442","rows_on_this_dataset":3,"code_links":2,"syntology":null},{"paper":"/paper/a-dual-modality-approach-for-zero-shot-multi","slug":"a-dual-modality-approach-for-zero-shot-multi","title":"Open Vocabulary Multi-Label Classification with Dual-Modal Decoder on Aligned Visual-Textual Features","date":"2022-08-19","arxiv_id":"2208.09562","rows_on_this_dataset":3,"code_links":0,"syntology":null},{"paper":"/paper/hierarchical-attention-network-for-few-shot","slug":"hierarchical-attention-network-for-few-shot","title":"Hierarchical Attention Network for Few-Shot Object Detection via Meta-Contrastive Learning","date":"2022-08-15","arxiv_id":"2208.07039","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/expansionnet-v2-block-static-expansion-in","slug":"expansionnet-v2-block-static-expansion-in","title":"Exploiting Multiple Sequence Lengths in Fast End to End Training for Image Captioning","date":"2022-08-13","arxiv_id":"2208.06551","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/analog-bits-generating-discrete-data-using","slug":"analog-bits-generating-discrete-data-using","title":"Analog Bits: Generating Discrete Data using Diffusion Models with Self-Conditioning","date":"2022-08-08","arxiv_id":"2208.04202","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":9,"samples_unverified":7,"pointer_only_for_licence":4,"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/analog-bits-generating-discrete-data-using#ran","syntology_url":"https://syntology.ai/paper/2208.04202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04202"}}}},{"paper":"/paper/aladin-distilling-fine-grained-alignment","slug":"aladin-distilling-fine-grained-alignment","title":"ALADIN: Distilling Fine-grained Alignment Scores for Efficient Image-Text Matching and Retrieval","date":"2022-07-29","arxiv_id":"2207.14757","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/hornet-efficient-high-order-spatial","slug":"hornet-efficient-high-order-spatial","title":"HorNet: Efficient High-Order Spatial Interactions with Recursive Gated Convolutions","date":"2022-07-28","arxiv_id":"2207.14284","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["raoyongming/hornet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hornet-efficient-high-order-spatial#ran","syntology_url":"https://syntology.ai/paper/2207.14284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.14284"}}}},{"paper":"/paper/k-means-mask-transformer","slug":"k-means-mask-transformer","title":"kMaX-DeepLab: k-means Mask Transformer","date":"2022-07-08","arxiv_id":"2207.04044","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/self-constrained-inference-optimization-on","slug":"self-constrained-inference-optimization-on","title":"Self-Constrained Inference Optimization on Structural Groups for Human Pose Estimation","date":"2022-07-06","arxiv_id":"2207.02425","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/yolov7-trainable-bag-of-freebies-sets-new","slug":"yolov7-trainable-bag-of-freebies-sets-new","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","date":"2022-07-06","arxiv_id":"2207.02696","rows_on_this_dataset":10,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":5,"official":{"repos":["wongkinyiu/yolov7"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov7-trainable-bag-of-freebies-sets-new#ran","syntology_url":"https://syntology.ai/paper/2207.02696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.02696"}}}},{"paper":"/paper/boosting-r-cnn-reweighting-r-cnn-samples-by","slug":"boosting-r-cnn-reweighting-r-cnn-samples-by","title":"Boosting R-CNN: Reweighting R-CNN Samples by RPN's Error for Underwater Object Detection","date":"2022-06-28","arxiv_id":"2206.13728","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/i-2r-net-intra-and-inter-human-relation","slug":"i-2r-net-intra-and-inter-human-relation","title":"I^2R-Net: Intra- and Inter-Human Relation Network for Multi-Person Pose Estimation","date":"2022-06-22","arxiv_id":"2206.10892","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/0-1-deep-neural-networks-via-block-coordinate","slug":"0-1-deep-neural-networks-via-block-coordinate","title":"0/1 Deep Neural Networks via Block Coordinate Descent","date":"2022-06-19","arxiv_id":"2206.09379","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cmt-deeplab-clustering-mask-transformers-for-1","slug":"cmt-deeplab-clustering-mask-transformers-for-1","title":"CMT-DeepLab: Clustering Mask Transformers for Panoptic Segmentation","date":"2022-06-17","arxiv_id":"2206.08948","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cmt-deeplab-clustering-mask-transformers-for-1#ran","syntology_url":"https://syntology.ai/paper/2206.08948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08948"}}}},{"paper":"/paper/deep-multi-task-networks-for-occluded","slug":"deep-multi-task-networks-for-occluded","title":"Deep Multi-Task Networks For Occluded Pedestrian Pose Estimation","date":"2022-06-15","arxiv_id":"2206.07510","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}}}},{"paper":"/paper/mask-dino-towards-a-unified-transformer-based-1","slug":"mask-dino-towards-a-unified-transformer-based-1","title":"Mask DINO: Towards A Unified Transformer-based Framework for Object Detection and Segmentation","date":"2022-06-06","arxiv_id":"2206.02777","rows_on_this_dataset":6,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":8,"samples_unverified":2,"pointer_only_for_licence":13,"official":{"repos":["idea-research/maskdino"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mask-dino-towards-a-unified-transformer-based-1#ran","syntology_url":"https://syntology.ai/paper/2206.02777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02777"}}}},{"paper":"/paper/architecture-agnostic-masked-image-modeling","slug":"architecture-agnostic-masked-image-modeling","title":"Architecture-Agnostic Masked Image Modeling -- From ViT back to CNN","date":"2022-05-27","arxiv_id":"2205.13943","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/contrastive-learning-rivals-masked-image","slug":"contrastive-learning-rivals-masked-image","title":"Contrastive Learning Rivals Masked Image Modeling in Fine-tuning via Feature Distillation","date":"2022-05-27","arxiv_id":"2205.14141","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":8,"official":{"repos":["SwinTransformer/Feature-Distillation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/contrastive-learning-rivals-masked-image#ran","syntology_url":"https://syntology.ai/paper/2205.14141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14141"}}}},{"paper":"/paper/mixmim-mixed-and-masked-image-modeling-for","slug":"mixmim-mixed-and-masked-image-modeling-for","title":"MixMAE: Mixed and Masked Autoencoder for Efficient Pretraining of Hierarchical Vision Transformers","date":"2022-05-26","arxiv_id":"2205.13137","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/revealing-the-dark-secrets-of-masked-image","slug":"revealing-the-dark-secrets-of-masked-image","title":"Revealing the Dark Secrets of Masked Image Modeling","date":"2022-05-26","arxiv_id":"2205.13543","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["SwinTransformer/MIM-Depth-Estimation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/revealing-the-dark-secrets-of-masked-image#ran","syntology_url":"https://syntology.ai/paper/2205.13543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13543"}}}},{"paper":"/paper/photorealistic-text-to-image-diffusion-models","slug":"photorealistic-text-to-image-diffusion-models","title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","date":"2022-05-23","arxiv_id":"2205.11487","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":5,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":6,"pointer_only_for_licence":9,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/photorealistic-text-to-image-diffusion-models#ran","syntology_url":"https://syntology.ai/paper/2205.11487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11487"}}}},{"paper":"/paper/uniform-masking-enabling-mae-pre-training-for","slug":"uniform-masking-enabling-mae-pre-training-for","title":"Uniform Masking: Enabling MAE Pre-training for Pyramid-based Vision Transformers with Locality","date":"2022-05-20","arxiv_id":"2205.10063","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/integral-migrating-pre-trained-transformer","slug":"integral-migrating-pre-trained-transformer","title":"Integrally Migrating Pre-trained Transformer Encoder-decoders for Visual Object Detection","date":"2022-05-19","arxiv_id":"2205.09613","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/vision-transformer-adapter-for-dense","slug":"vision-transformer-adapter-for-dense","title":"Vision Transformer Adapter for Dense Predictions","date":"2022-05-17","arxiv_id":"2205.08534","rows_on_this_dataset":11,"code_links":2,"syntology":null},{"paper":"/paper/simple-open-vocabulary-object-detection-with","slug":"simple-open-vocabulary-object-detection-with","title":"Simple Open-Vocabulary Object Detection with Vision Transformers","date":"2022-05-12","arxiv_id":"2205.06230","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/aggpose-deep-aggregation-vision-transformer","slug":"aggpose-deep-aggregation-vision-transformer","title":"AggPose: Deep Aggregation Vision Transformer for Infant Pose Estimation","date":"2022-05-11","arxiv_id":"2205.05277","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":5,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":7,"official":{"repos":["szar-lab/aggpose"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/aggpose-deep-aggregation-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.05277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05277"}}}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":10,"samples_constructed":5,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":7,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}}}},{"paper":"/paper/lite-pose-efficient-architecture-design-for","slug":"lite-pose-efficient-architecture-design-for","title":"Lite Pose: Efficient Architecture Design for 2D Human Pose Estimation","date":"2022-05-03","arxiv_id":"2205.01271","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["mit-han-lab/litepose"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lite-pose-efficient-architecture-design-for#ran","syntology_url":"https://syntology.ai/paper/2205.01271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01271"}}}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":24,"samples_ran":18,"samples_constructed":6,"samples_ran_checked":12,"samples_ran_instrument_failed":6,"samples_unverified":6,"pointer_only_for_licence":8,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}}}},{"paper":"/paper/cogview2-faster-and-better-text-to-image","slug":"cogview2-faster-and-better-text-to-image","title":"CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers","date":"2022-04-28","arxiv_id":"2204.14217","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["thudm/cogview2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogview2-faster-and-better-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2204.14217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14217"}}}},{"paper":"/paper/understanding-the-robustness-in-vision","slug":"understanding-the-robustness-in-vision","title":"Understanding The Robustness in Vision Transformers","date":"2022-04-26","arxiv_id":"2204.12451","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/vitpose-simple-vision-transformer-baselines","slug":"vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","arxiv_id":"2204.12484","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":31,"samples_ran":18,"samples_constructed":8,"samples_ran_checked":12,"samples_ran_instrument_failed":6,"samples_unverified":13,"pointer_only_for_licence":6,"official":{"repos":["vitae-transformer/vitpose"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vitpose-simple-vision-transformer-baselines#ran","syntology_url":"https://syntology.ai/paper/2204.12484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12484"}}}},{"paper":"/paper/dite-hrnet-dynamic-lightweight-high","slug":"dite-hrnet-dynamic-lightweight-high","title":"Dite-HRNet: Dynamic Lightweight High-Resolution Network for Human Pose Estimation","date":"2022-04-22","arxiv_id":"2204.10762","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/recurrent-affine-transformation-for-text-to","slug":"recurrent-affine-transformation-for-text-to","title":"Recurrent Affine Transformation for Text-to-image Synthesis","date":"2022-04-22","arxiv_id":"2204.10482","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/centernet-for-object-detection","slug":"centernet-for-object-detection","title":"CenterNet++ for Object Detection","date":"2022-04-18","arxiv_id":"2204.08394","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/yolo-pose-enhancing-yolo-for-multi-person","slug":"yolo-pose-enhancing-yolo-for-multi-person","title":"YOLO-Pose: Enhancing YOLO for Multi Person Pose Estimation Using Object Keypoint Similarity Loss","date":"2022-04-14","arxiv_id":"2204.06806","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/hierarchical-text-conditional-image","slug":"hierarchical-text-conditional-image","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","date":"2022-04-13","arxiv_id":"2204.06125","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":38,"samples_ran":33,"samples_constructed":7,"samples_ran_checked":26,"samples_ran_instrument_failed":7,"samples_unverified":5,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hierarchical-text-conditional-image#ran","syntology_url":"https://syntology.ai/paper/2204.06125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06125"}}}},{"paper":"/paper/cfa-constraint-based-finetuning-approach-for","slug":"cfa-constraint-based-finetuning-approach-for","title":"CFA: Constraint-based Finetuning Approach for Generalized Few-Shot Object Detection","date":"2022-04-11","arxiv_id":"2204.05220","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/davit-dual-attention-vision-transformers","slug":"davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","arxiv_id":"2204.03645","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":8,"samples_constructed":5,"samples_ran_checked":6,"samples_ran_instrument_failed":2,"samples_unverified":7,"pointer_only_for_licence":0,"official":{"repos":["dingmyu/davit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/davit-dual-attention-vision-transformers#ran","syntology_url":"https://syntology.ai/paper/2204.03645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.03645"}}}},{"paper":"/paper/knn-diffusion-image-generation-via-large","slug":"knn-diffusion-image-generation-via-large","title":"KNN-Diffusion: Image Generation via Large-Scale Retrieval","date":"2022-04-06","arxiv_id":"2204.02849","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/maxvit-multi-axis-vision-transformer","slug":"maxvit-multi-axis-vision-transformer","title":"MaxViT: Multi-Axis Vision Transformer","date":"2022-04-04","arxiv_id":"2204.01697","rows_on_this_dataset":3,"code_links":15,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":53,"samples_ran":37,"samples_constructed":15,"samples_ran_checked":23,"samples_ran_instrument_failed":14,"samples_unverified":16,"pointer_only_for_licence":9,"official":{"repos":["google-research/maxvit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/maxvit-multi-axis-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2204.01697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01697"}}}},{"paper":"/paper/vista-vision-and-scene-text-aggregation-for","slug":"vista-vision-and-scene-text-aggregation-for","title":"ViSTA: Vision and Scene Text Aggregation for Cross-Modal Retrieval","date":"2022-03-31","arxiv_id":"2203.16778","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/exploring-plain-vision-transformer-backbones","slug":"exploring-plain-vision-transformer-backbones","title":"Exploring Plain Vision Transformer Backbones for Object Detection","date":"2022-03-30","arxiv_id":"2203.16527","rows_on_this_dataset":4,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["facebookresearch/detectron2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/exploring-plain-vision-transformer-backbones#ran","syntology_url":"https://syntology.ai/paper/2203.16527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16527"}}}},{"paper":"/paper/pp-yoloe-an-evolved-version-of-yolo","slug":"pp-yoloe-an-evolved-version-of-yolo","title":"PP-YOLOE: An evolved version of YOLO","date":"2022-03-30","arxiv_id":"2203.16250","rows_on_this_dataset":9,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":27,"samples_ran":22,"samples_constructed":0,"samples_ran_checked":21,"samples_ran_instrument_failed":1,"samples_unverified":5,"pointer_only_for_licence":1,"official":{"repos":["PaddlePaddle/PaddleDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/pp-yoloe-an-evolved-version-of-yolo#ran","syntology_url":"https://syntology.ai/paper/2203.16250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16250"}}}},{"paper":"/paper/make-a-scene-scene-based-text-to-image","slug":"make-a-scene-scene-based-text-to-image","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","date":"2022-03-24","arxiv_id":"2203.13131","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":2,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/make-a-scene-scene-based-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2203.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13131"}}}},{"paper":"/paper/focal-modulation-networks","slug":"focal-modulation-networks","title":"Focal Modulation Networks","date":"2022-03-22","arxiv_id":"2203.11926","rows_on_this_dataset":5,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/FocalNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/focal-modulation-networks#ran","syntology_url":"https://syntology.ai/paper/2203.11926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11926"}}}},{"paper":"/paper/activemlp-an-mlp-like-architecture-with","slug":"activemlp-an-mlp-like-architecture-with","title":"Active Token Mixer","date":"2022-03-11","arxiv_id":"2203.06108","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/e2ec-an-end-to-end-contour-based-method-for","slug":"e2ec-an-end-to-end-contour-based-method-for","title":"E2EC: An End-to-End Contour-based Method for High-Quality High-Speed Instance Segmentation","date":"2022-03-08","arxiv_id":"2203.04074","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":10,"official":{"repos":["zhang-tao-whu/e2ec"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/e2ec-an-end-to-end-contour-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2203.04074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.04074"}}}},{"paper":"/paper/dino-detr-with-improved-denoising-anchor-1","slug":"dino-detr-with-improved-denoising-anchor-1","title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","date":"2022-03-07","arxiv_id":"2203.03605","rows_on_this_dataset":4,"code_links":16,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":4,"samples_unverified":5,"pointer_only_for_licence":5,"official":{"repos":["IDEACVR/DINO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dino-detr-with-improved-denoising-anchor-1#ran","syntology_url":"https://syntology.ai/paper/2203.03605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03605"}}}},{"paper":"/paper/dn-detr-accelerate-detr-training-by","slug":"dn-detr-accelerate-detr-training-by","title":"DN-DETR: Accelerate DETR Training by Introducing Query DeNoising","date":"2022-03-02","arxiv_id":"2203.01305","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":22,"samples_ran":17,"samples_constructed":3,"samples_ran_checked":6,"samples_ran_instrument_failed":11,"samples_unverified":5,"pointer_only_for_licence":9,"official":{"repos":["IDEA-Research/detrex","fengli-ust/dn-detr","idea-research/dn-detr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dn-detr-accelerate-detr-training-by#ran","syntology_url":"https://syntology.ai/paper/2203.01305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01305"}}}},{"paper":"/paper/lile-look-in-depth-before-looking-elsewhere-a","slug":"lile-look-in-depth-before-looking-elsewhere-a","title":"LILE: Look In-Depth before Looking Elsewhere -- A Dual Attention Network using Transformers for Cross-Modal Information Retrieval in Histopathology Archives","date":"2022-03-02","arxiv_id":"2203.01445","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/isda-position-aware-instance-segmentation","slug":"isda-position-aware-instance-segmentation","title":"ISDA: Position-Aware Instance Segmentation with Deformable Attention","date":"2022-02-23","arxiv_id":"2202.12251","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/self-supervised-transformers-for-unsupervised","slug":"self-supervised-transformers-for-unsupervised","title":"Self-Supervised Transformers for Unsupervised Object Discovery using Normalized Cut","date":"2022-02-23","arxiv_id":"2202.11539","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-transformers-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2202.11539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.11539"}}}}],"record_sha256":"7baa7be0efd003b1540782b2945a19844907895cb6a2f30844e6413801a44e01","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}