{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/coco/papers/3","list_of":"/dataset/coco","dataset":"COCO (Common Objects in Context)","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"archive","order_definition":"date (newest first), then slug","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":3,"pages_in_order":6,"rows_per_page":100,"rows":[201,300],"of":579,"counts":{"papers_with_a_benchmark_row":579,"with_a_code_link":504,"where_syntology_ran_a_sample":256,"not_listed_spam_title":0,"listed":579,"listed_where_code_ran":256,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":226,"every_run_a_failure_of_syntologys_instrument":30,"listed_with_a_run_with_no_instrument_failure":226,"listed_every_run_a_failure_of_syntologys_instrument":30,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/coco","prev":"/dataset/coco/papers/2","next":"/dataset/coco/papers/4","papers":[{"paper":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}}}},{"paper":"/paper/visual-attention-network","slug":"visual-attention-network","title":"Visual Attention Network","date":"2022-02-20","arxiv_id":"2202.09741","rows_on_this_dataset":2,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["Visual-Attention-Network/VAN-Classification"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/visual-attention-network#ran","syntology_url":"https://syntology.ai/paper/2202.09741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09741"}}}},{"paper":"/paper/truncated-diffusion-probabilistic-models","slug":"truncated-diffusion-probabilistic-models","title":"Truncated Diffusion Probabilistic Models and Diffusion-based Adversarial Auto-Encoders","date":"2022-02-19","arxiv_id":"2202.09671","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["jegzheng/truncated-diffusion-probabilistic-models"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/truncated-diffusion-probabilistic-models#ran","syntology_url":"https://syntology.ai/paper/2202.09671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09671"}}}},{"paper":"/paper/from-node-to-graph-joint-reasoning-on-visual","slug":"from-node-to-graph-joint-reasoning-on-visual","title":"From Node to Graph: Joint Reasoning on Visual-Semantic Relational Graph for Zero-Shot Detection","date":"2022-02-15","arxiv_id":null,"rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/context-autoencoder-for-self-supervised","slug":"context-autoencoder-for-self-supervised","title":"Context Autoencoder for Self-Supervised Representation Learning","date":"2022-02-07","arxiv_id":"2202.03026","rows_on_this_dataset":1,"code_links":6,"syntology":null},{"paper":"/paper/the-devil-is-in-the-labels-semantic","slug":"the-devil-is-in-the-labels-semantic","title":"Scaling up Multi-domain Semantic Segmentation with Sentence Embeddings","date":"2022-02-04","arxiv_id":"2202.02002","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dab-detr-dynamic-anchor-boxes-are-better-1","slug":"dab-detr-dynamic-anchor-boxes-are-better-1","title":"DAB-DETR: Dynamic Anchor Boxes are Better Queries for DETR","date":"2022-01-28","arxiv_id":"2201.12329","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":6,"samples_constructed":3,"samples_ran_checked":4,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":0,"official":{"repos":["slongliu/dab-detr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dab-detr-dynamic-anchor-boxes-are-better-1#ran","syntology_url":"https://syntology.ai/paper/2201.12329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12329"}}}},{"paper":"/paper/when-shift-operation-meets-vision-transformer","slug":"when-shift-operation-meets-vision-transformer","title":"When Shift Operation Meets Vision Transformer: An Extremely Simple Alternative to Attention Mechanism","date":"2022-01-26","arxiv_id":"2201.10801","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":6,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["microsoft/SPACH"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/when-shift-operation-meets-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2201.10801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.10801"}}}},{"paper":"/paper/poseur-direct-human-pose-regression-with","slug":"poseur-direct-human-pose-regression-with","title":"Poseur: Direct Human Pose Regression with Transformers","date":"2022-01-19","arxiv_id":"2201.07412","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-with-deformable-attention","slug":"vision-transformer-with-deformable-attention","title":"Vision Transformer with Deformable Attention","date":"2022-01-03","arxiv_id":"2201.00520","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":4,"samples_ran_checked":4,"samples_ran_instrument_failed":4,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["leaplabthu/dat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vision-transformer-with-deformable-attention#ran","syntology_url":"https://syntology.ai/paper/2201.00520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.00520"}}}},{"paper":"/paper/contextual-debiasing-for-visual-recognition","slug":"contextual-debiasing-for-visual-recognition","title":"Contextual Debiasing for Visual Recognition With Causal Mechanisms","date":"2022-01-01","arxiv_id":null,"rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/robust-region-feature-synthesizer-for-zero","slug":"robust-region-feature-synthesizer-for-zero","title":"Robust Region Feature Synthesizer for Zero-Shot Object Detection","date":"2022-01-01","arxiv_id":"2201.00103","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["HPL123/RRFS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/robust-region-feature-synthesizer-for-zero#ran","syntology_url":"https://syntology.ai/paper/2201.00103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.00103"}}}},{"paper":"/paper/ernie-vilg-unified-generative-pre-training","slug":"ernie-vilg-unified-generative-pre-training","title":"ERNIE-ViLG: Unified Generative Pre-training for Bidirectional Vision-Language Generation","date":"2021-12-31","arxiv_id":"2112.15283","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/augmenting-convolutional-networks-with","slug":"augmenting-convolutional-networks-with","title":"Augmenting Convolutional networks with attention-based aggregation","date":"2021-12-27","arxiv_id":"2112.13692","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["facebookresearch/deit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/augmenting-convolutional-networks-with#ran","syntology_url":"https://syntology.ai/paper/2112.13692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.13692"}}}},{"paper":"/paper/elsa-enhanced-local-self-attention-for-vision","slug":"elsa-enhanced-local-self-attention-for-vision","title":"ELSA: Enhanced Local Self-Attention for Vision Transformer","date":"2021-12-23","arxiv_id":"2112.12786","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["damo-cv/elsa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/elsa-enhanced-local-self-attention-for-vision#ran","syntology_url":"https://syntology.ai/paper/2112.12786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12786"}}}},{"paper":"/paper/mpvit-multi-path-vision-transformer-for-dense","slug":"mpvit-multi-path-vision-transformer-for-dense","title":"MPViT: Multi-Path Vision Transformer for Dense Prediction","date":"2021-12-21","arxiv_id":"2112.11010","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/bapose-bottom-up-pose-estimation-with","slug":"bapose-bottom-up-pose-estimation-with","title":"BAPose: Bottom-Up Pose Estimation with Disentangled Waterfall Representations","date":"2021-12-20","arxiv_id":"2112.10716","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/glide-towards-photorealistic-image-generation","slug":"glide-towards-photorealistic-image-generation","title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models","date":"2021-12-20","arxiv_id":"2112.10741","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":9,"samples_constructed":7,"samples_ran_checked":8,"samples_ran_instrument_failed":1,"samples_unverified":6,"pointer_only_for_licence":0,"official":{"repos":["openai/glide-text2im"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/glide-towards-photorealistic-image-generation#ran","syntology_url":"https://syntology.ai/paper/2112.10741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10741"}}}},{"paper":"/paper/high-resolution-image-synthesis-with-latent","slug":"high-resolution-image-synthesis-with-latent","title":"High-Resolution Image Synthesis with Latent Diffusion Models","date":"2021-12-20","arxiv_id":"2112.10752","rows_on_this_dataset":3,"code_links":41,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":28,"samples_ran":22,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":12,"samples_unverified":6,"pointer_only_for_licence":10,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/high-resolution-image-synthesis-with-latent#ran","syntology_url":"https://syntology.ai/paper/2112.10752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10752"}}}},{"paper":"/paper/pe-former-pose-estimation-transformer","slug":"pe-former-pose-estimation-transformer","title":"PE-former: Pose Estimation Transformer","date":"2021-12-09","arxiv_id":"2112.04981","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/recurrent-glimpse-based-decoder-for-detection","slug":"recurrent-glimpse-based-decoder-for-detection","title":"Recurrent Glimpse-based Decoder for Detection with Transformer","date":"2021-12-09","arxiv_id":"2112.04632","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/flava-a-foundational-language-and-vision","slug":"flava-a-foundational-language-and-vision","title":"FLAVA: A Foundational Language And Vision Alignment Model","date":"2021-12-08","arxiv_id":"2112.04482","rows_on_this_dataset":3,"code_links":4,"syntology":null},{"paper":"/paper/grounded-language-image-pre-training","slug":"grounded-language-image-pre-training","title":"Grounded Language-Image Pre-training","date":"2021-12-07","arxiv_id":"2112.03857","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/grounded-language-image-pre-training#ran","syntology_url":"https://syntology.ai/paper/2112.03857","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03857"}}}},{"paper":"/paper/fusedream-training-free-text-to-image","slug":"fusedream-training-free-text-to-image","title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","date":"2021-12-02","arxiv_id":"2112.01573","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/improved-multiscale-vision-transformers-for","slug":"improved-multiscale-vision-transformers-for","title":"MViTv2: Improved Multiscale Vision Transformers for Classification and Detection","date":"2021-12-02","arxiv_id":"2112.01526","rows_on_this_dataset":8,"code_links":9,"syntology":null},{"paper":"/paper/masked-attention-mask-transformer-for","slug":"masked-attention-mask-transformer-for","title":"Masked-attention Mask Transformer for Universal Image Segmentation","date":"2021-12-02","arxiv_id":"2112.01527","rows_on_this_dataset":6,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["facebookresearch/Mask2Former"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/masked-attention-mask-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2112.01527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01527"}}}},{"paper":"/paper/vector-quantized-diffusion-model-for-text-to","slug":"vector-quantized-diffusion-model-for-text-to","title":"Vector Quantized Diffusion Model for Text-to-Image Synthesis","date":"2021-11-29","arxiv_id":"2111.14822","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/lafite-towards-language-free-training-for","slug":"lafite-towards-language-free-training-for","title":"LAFITE: Towards Language-Free Training for Text-to-Image Generation","date":"2021-11-27","arxiv_id":"2111.13792","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":2,"samples_unverified":6,"pointer_only_for_licence":3,"official":{"repos":["drboog/Lafite"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lafite-towards-language-free-training-for#ran","syntology_url":"https://syntology.ai/paper/2111.13792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13792"}}}},{"paper":"/paper/mask-transfiner-for-high-quality-instance","slug":"mask-transfiner-for-high-quality-instance","title":"Mask Transfiner for High-Quality Instance Segmentation","date":"2021-11-26","arxiv_id":"2111.13673","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["SysCV/transfiner"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mask-transfiner-for-high-quality-instance#ran","syntology_url":"https://syntology.ai/paper/2111.13673","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.13673"}}}},{"paper":"/paper/revisiting-efficient-object-detection","slug":"revisiting-efficient-object-detection","title":"MAE-DET: Revisiting Maximum Entropy Principle in Zero-Shot NAS for Efficient Object Detection","date":"2021-11-26","arxiv_id":"2111.13336","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/attend-to-who-you-are-supervising-self-1","slug":"attend-to-who-you-are-supervising-self-1","title":"Attend to Who You Are: Supervising Self-Attention for Keypoint Detection and Instance-Aware Association","date":"2021-11-25","arxiv_id":"2111.12892","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/ml-decoder-scalable-and-versatile","slug":"ml-decoder-scalable-and-versatile","title":"ML-Decoder: Scalable and Versatile Classification Head","date":"2021-11-25","arxiv_id":"2111.12933","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["alibaba-miil/ml_decoder"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ml-decoder-scalable-and-versatile#ran","syntology_url":"https://syntology.ai/paper/2111.12933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12933"}}}},{"paper":"/paper/nuwa-visual-synthesis-pre-training-for-neural","slug":"nuwa-visual-synthesis-pre-training-for-neural","title":"NÜWA: Visual Synthesis Pre-training for Neural visUal World creAtion","date":"2021-11-24","arxiv_id":"2111.12417","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/nuwa-visual-synthesis-pre-training-for-neural#ran","syntology_url":"https://syntology.ai/paper/2111.12417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12417"}}}},{"paper":"/paper/focal-and-global-knowledge-distillation-for","slug":"focal-and-global-knowledge-distillation-for","title":"Focal and Global Knowledge Distillation for Detectors","date":"2021-11-23","arxiv_id":"2111.11837","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/florence-a-new-foundation-model-for-computer","slug":"florence-a-new-foundation-model-for-computer","title":"Florence: A New Foundation Model for Computer Vision","date":"2021-11-22","arxiv_id":"2111.11432","rows_on_this_dataset":4,"code_links":2,"syntology":null},{"paper":"/paper/l-verse-bidirectional-generation-between","slug":"l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","arxiv_id":"2111.11133","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/metaformer-is-actually-what-you-need-for","slug":"metaformer-is-actually-what-you-need-for","title":"MetaFormer Is Actually What You Need for Vision","date":"2021-11-22","arxiv_id":"2111.11418","rows_on_this_dataset":1,"code_links":18,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["rwightman/pytorch-image-models","sail-sg/poolformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/metaformer-is-actually-what-you-need-for#ran","syntology_url":"https://syntology.ai/paper/2111.11418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.11418"}}}},{"paper":"/paper/multi-modal-transformers-excel-at-class","slug":"multi-modal-transformers-excel-at-class","title":"Class-agnostic Object Detection with Multi-modal Transformer","date":"2021-11-22","arxiv_id":"2111.11430","rows_on_this_dataset":4,"code_links":1,"syntology":null},{"paper":"/paper/swin-transformer-v2-scaling-up-capacity-and","slug":"swin-transformer-v2-scaling-up-capacity-and","title":"Swin Transformer V2: Scaling Up Capacity and Resolution","date":"2021-11-18","arxiv_id":"2111.09883","rows_on_this_dataset":4,"code_links":23,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":30,"samples_ran":17,"samples_constructed":0,"samples_ran_checked":17,"samples_ran_instrument_failed":0,"samples_unverified":13,"pointer_only_for_licence":4,"official":{"repos":["microsoft/Swin-Transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/swin-transformer-v2-scaling-up-capacity-and#ran","syntology_url":"https://syntology.ai/paper/2111.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09883"}}}},{"paper":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}}}},{"paper":"/paper/rethinking-keypoint-representations-modeling","slug":"rethinking-keypoint-representations-modeling","title":"Rethinking Keypoint Representations: Modeling Keypoints and Poses as Objects for Multi-Person Human Pose Estimation","date":"2021-11-16","arxiv_id":"2111.08557","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/ibot-image-bert-pre-training-with-online","slug":"ibot-image-bert-pre-training-with-online","title":"iBOT: Image BERT Pre-Training with Online Tokenizer","date":"2021-11-15","arxiv_id":"2111.07832","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["bytedance/ibot"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ibot-image-bert-pre-training-with-online#ran","syntology_url":"https://syntology.ai/paper/2111.07832","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.07832"}}}},{"paper":"/paper/masked-autoencoders-are-scalable-vision","slug":"masked-autoencoders-are-scalable-vision","title":"Masked Autoencoders Are Scalable Vision Learners","date":"2021-11-11","arxiv_id":"2111.06377","rows_on_this_dataset":2,"code_links":58,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":137,"samples_ran":86,"samples_constructed":40,"samples_ran_checked":69,"samples_ran_instrument_failed":17,"samples_unverified":51,"pointer_only_for_licence":78,"official":{"repos":["facebookresearch/mae"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/masked-autoencoders-are-scalable-vision#ran","syntology_url":"https://syntology.ai/paper/2111.06377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.06377"}}}},{"paper":"/paper/an-empirical-study-of-training-end-to-end","slug":"an-empirical-study-of-training-end-to-end","title":"An Empirical Study of Training End-to-End Vision-and-Language Transformers","date":"2021-11-03","arxiv_id":"2111.02387","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["zdou0830/meter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-empirical-study-of-training-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2111.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02387"}}}},{"paper":"/paper/plug-and-play-few-shot-object-detection-with","slug":"plug-and-play-few-shot-object-detection-with","title":"Instant Response Few-shot Object Detection with Meta Strategy and Explicit Localization Inference","date":"2021-10-26","arxiv_id":"2110.13377","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hrformer-high-resolution-transformer-for","slug":"hrformer-high-resolution-transformer-for","title":"HRFormer: High-Resolution Transformer for Dense Prediction","date":"2021-10-18","arxiv_id":"2110.09408","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":7,"samples_constructed":7,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":0,"official":{"repos":["HRNet/HRFormer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hrformer-high-resolution-transformer-for#ran","syntology_url":"https://syntology.ai/paper/2110.09408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.09408"}}}},{"paper":"/paper/the-center-of-attention-center-keypoint-1","slug":"the-center-of-attention-center-keypoint-1","title":"The Center of Attention: Center-Keypoint Grouping via Attention for Multi-Person Pose Estimation","date":"2021-10-11","arxiv_id":"2110.05132","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/transformer-based-dual-relation-graph-for-1","slug":"transformer-based-dual-relation-graph-for-1","title":"Transformer-based Dual Relation Graph for Multi-label Image Recognition","date":"2021-10-10","arxiv_id":"2110.04722","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":9,"samples_constructed":6,"samples_ran_checked":8,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["iCVTEAM/TDRG"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/transformer-based-dual-relation-graph-for-1#ran","syntology_url":"https://syntology.ai/paper/2110.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04722"}}}},{"paper":"/paper/infoseg-unsupervised-semantic-image","slug":"infoseg-unsupervised-semantic-image","title":"InfoSeg: Unsupervised Semantic Image Segmentation with Mutual Information Maximization","date":"2021-10-07","arxiv_id":"2110.03477","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/m3tr-multi-modal-multi-label-recognition-with","slug":"m3tr-multi-modal-multi-label-recognition-with","title":"M3TR: Multi-modal Multi-label Recognition with Transformer","date":"2021-10-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/localizing-objects-with-self-supervised","slug":"localizing-objects-with-self-supervised","title":"Localizing Objects with Self-Supervised Transformers and no Labels","date":"2021-09-29","arxiv_id":"2109.14279","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/pix2seq-a-language-modeling-framework-for","slug":"pix2seq-a-language-modeling-framework-for","title":"Pix2seq: A Language Modeling Framework for Object Detection","date":"2021-09-22","arxiv_id":"2109.10852","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":10,"samples_constructed":3,"samples_ran_checked":6,"samples_ran_instrument_failed":4,"samples_unverified":2,"pointer_only_for_licence":10,"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/pix2seq-a-language-modeling-framework-for#ran","syntology_url":"https://syntology.ai/paper/2109.10852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10852"}}}},{"paper":"/paper/beyond-semantic-to-instance-segmentation","slug":"beyond-semantic-to-instance-segmentation","title":"Beyond Semantic to Instance Segmentation: Weakly-Supervised Instance Segmentation via Semantic Knowledge Transfer and Self-Refinement","date":"2021-09-20","arxiv_id":"2109.09477","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hptq-hardware-friendly-post-training","slug":"hptq-hardware-friendly-post-training","title":"HPTQ: Hardware-Friendly Post Training Quantization","date":"2021-09-19","arxiv_id":"2109.09113","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/few-shot-object-detection-by-attending-to-per","slug":"few-shot-object-detection-by-attending-to-per","title":"Few-Shot Object Detection by Attending to Per-Sample-Prototype","date":"2021-09-16","arxiv_id":"2109.07734","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/anchor-detr-query-design-for-transformer","slug":"anchor-detr-query-design-for-transformer","title":"Anchor DETR: Query Design for Transformer-Based Object Detection","date":"2021-09-15","arxiv_id":"2109.07107","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/panoptic-segformer","slug":"panoptic-segformer","title":"Panoptic SegFormer: Delving Deeper into Panoptic Segmentation with Transformers","date":"2021-09-08","arxiv_id":"2109.03814","rows_on_this_dataset":6,"code_links":3,"syntology":null},{"paper":"/paper/semantics-guided-contrastive-network-for-zero","slug":"semantics-guided-contrastive-network-for-zero","title":"Semantics-Guided Contrastive Network for Zero-Shot Object detection","date":"2021-09-04","arxiv_id":"2109.06062","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/defrcn-decoupled-faster-r-cnn-for-few-shot","slug":"defrcn-decoupled-faster-r-cnn-for-few-shot","title":"DeFRCN: Decoupled Faster R-CNN for Few-Shot Object Detection","date":"2021-08-20","arxiv_id":"2108.09017","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["er-muyue/defrcn"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/defrcn-decoupled-faster-r-cnn-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2108.09017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.09017"}}}},{"paper":"/paper/perturb-predict-paraphrase-semi-supervised","slug":"perturb-predict-paraphrase-semi-supervised","title":"Perturb, Predict & Paraphrase: Semi-Supervised Learning using Noisy Student for Image Captioning","date":"2021-08-19","arxiv_id":null,"rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/tood-task-aligned-one-stage-object-detection","slug":"tood-task-aligned-one-stage-object-detection","title":"TOOD: Task-aligned One-stage Object Detection","date":"2021-08-17","arxiv_id":"2108.07755","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["fcjian/TOOD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tood-task-aligned-one-stage-object-detection#ran","syntology_url":"https://syntology.ai/paper/2108.07755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.07755"}}}},{"paper":"/paper/conditional-detr-for-fast-training","slug":"conditional-detr-for-fast-training","title":"Conditional DETR for Fast Training Convergence","date":"2021-08-13","arxiv_id":"2108.06152","rows_on_this_dataset":4,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":1,"samples_ran_checked":4,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":4,"official":{"repos":["atten4vis/conditionaldetr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/conditional-detr-for-fast-training#ran","syntology_url":"https://syntology.ai/paper/2108.06152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.06152"}}}},{"paper":"/paper/paint-transformer-feed-forward-neural","slug":"paint-transformer-feed-forward-neural","title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","date":"2021-08-09","arxiv_id":"2108.03798","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/rethinking-and-improving-relative-position","slug":"rethinking-and-improving-relative-position","title":"Rethinking and Improving Relative Position Encoding for Vision Transformer","date":"2021-07-29","arxiv_id":"2107.14222","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":7,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["microsoft/cream"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rethinking-and-improving-relative-position#ran","syntology_url":"https://syntology.ai/paper/2107.14222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.14222"}}}},{"paper":"/paper/query2label-a-simple-transformer-way-to-multi","slug":"query2label-a-simple-transformer-way-to-multi","title":"Query2Label: A Simple Transformer Way to Multi-Label Classification","date":"2021-07-22","arxiv_id":"2107.10834","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["SlongLiu/query2labels"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/query2label-a-simple-transformer-way-to-multi#ran","syntology_url":"https://syntology.ai/paper/2107.10834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.10834"}}}},{"paper":"/paper/inspose-instance-aware-networks-for-single","slug":"inspose-instance-aware-networks-for-single","title":"InsPose: Instance-Aware Networks for Single-Stage Multi-Person Pose Estimation","date":"2021-07-19","arxiv_id":"2107.08982","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/yolox-exceeding-yolo-series-in-2021","slug":"yolox-exceeding-yolo-series-in-2021","title":"YOLOX: Exceeding YOLO Series in 2021","date":"2021-07-18","arxiv_id":"2107.08430","rows_on_this_dataset":4,"code_links":42,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":23,"samples_ran":16,"samples_constructed":0,"samples_ran_checked":15,"samples_ran_instrument_failed":1,"samples_unverified":7,"pointer_only_for_licence":0,"official":{"repos":["Megvii-BaseDetection/YOLOX"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolox-exceeding-yolo-series-in-2021#ran","syntology_url":"https://syntology.ai/paper/2107.08430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.08430"}}}},{"paper":"/paper/align-before-fuse-vision-and-language","slug":"align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","arxiv_id":"2107.07651","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/align-before-fuse-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2107.07651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07651"}}}},{"paper":"/paper/per-pixel-classification-is-not-all-you-need","slug":"per-pixel-classification-is-not-all-you-need","title":"Per-Pixel Classification is Not All You Need for Semantic Segmentation","date":"2021-07-13","arxiv_id":"2107.06278","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/greedy-offset-guided-keypoint-grouping-for","slug":"greedy-offset-guided-keypoint-grouping-for","title":"Greedy Offset-Guided Keypoint Grouping for Human Pose Estimation","date":"2021-07-07","arxiv_id":"2107.03098","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/improving-text-to-image-synthesis-using","slug":"improving-text-to-image-synthesis-using","title":"Improving Text-to-Image Synthesis Using Contrastive Learning","date":"2021-07-06","arxiv_id":"2107.02423","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/polarized-self-attention-towards-high-quality-1","slug":"polarized-self-attention-towards-high-quality-1","title":"Polarized Self-Attention: Towards High-quality Pixel-wise Regression","date":"2021-07-02","arxiv_id":"2107.00782","rows_on_this_dataset":3,"code_links":6,"syntology":null},{"paper":"/paper/cbnetv2-a-composite-backbone-network","slug":"cbnetv2-a-composite-backbone-network","title":"CBNet: A Composite Backbone Network Architecture for Object Detection","date":"2021-07-01","arxiv_id":"2107.00420","rows_on_this_dataset":9,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["VDIGPKU/CBNetV2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cbnetv2-a-composite-backbone-network#ran","syntology_url":"https://syntology.ai/paper/2107.00420","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00420"}}}},{"paper":"/paper/focal-self-attention-for-local-global","slug":"focal-self-attention-for-local-global","title":"Focal Self-attention for Local-Global Interactions in Vision Transformers","date":"2021-07-01","arxiv_id":"2107.00641","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["microsoft/Focal-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/focal-self-attention-for-local-global#ran","syntology_url":"https://syntology.ai/paper/2107.00641","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.00641"}}}},{"paper":"/paper/unsupervised-image-segmentation-by-mutual","slug":"unsupervised-image-segmentation-by-mutual","title":"Unsupervised Image Segmentation by Mutual Information Maximization and Adversarial Regularization","date":"2021-07-01","arxiv_id":"2107.00691","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/simple-training-strategies-and-model-scaling","slug":"simple-training-strategies-and-model-scaling","title":"Simple Training Strategies and Model Scaling for Object Detection","date":"2021-06-30","arxiv_id":"2107.00057","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/k-net-towards-unified-image-segmentation","slug":"k-net-towards-unified-image-segmentation","title":"K-Net: Towards Unified Image Segmentation","date":"2021-06-28","arxiv_id":"2106.14855","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["zwwwayne/k-net"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/k-net-towards-unified-image-segmentation#ran","syntology_url":"https://syntology.ai/paper/2106.14855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14855"}}}},{"paper":"/paper/pvtv2-improved-baselines-with-pyramid-vision","slug":"pvtv2-improved-baselines-with-pyramid-vision","title":"PVT v2: Improved Baselines with Pyramid Vision Transformer","date":"2021-06-25","arxiv_id":"2106.13797","rows_on_this_dataset":1,"code_links":18,"syntology":null},{"paper":"/paper/multi-layered-semantic-representation-network","slug":"multi-layered-semantic-representation-network","title":"Multi-layered Semantic Representation Network for Multi-label Image Classification","date":"2021-06-22","arxiv_id":"2106.11596","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/xcit-cross-covariance-image-transformers","slug":"xcit-cross-covariance-image-transformers","title":"XCiT: Cross-Covariance Image Transformers","date":"2021-06-17","arxiv_id":"2106.09681","rows_on_this_dataset":4,"code_links":12,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":11,"official":{"repos":["facebookresearch/xcit","rwightman/pytorch-image-models"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/xcit-cross-covariance-image-transformers#ran","syntology_url":"https://syntology.ai/paper/2106.09681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.09681"}}}},{"paper":"/paper/end-to-end-semi-supervised-object-detection","slug":"end-to-end-semi-supervised-object-detection","title":"End-to-End Semi-Supervised Object Detection with Soft Teacher","date":"2021-06-16","arxiv_id":"2106.09018","rows_on_this_dataset":6,"code_links":8,"syntology":null},{"paper":"/paper/dynamic-head-unifying-object-detection-heads","slug":"dynamic-head-unifying-object-detection-heads","title":"Dynamic Head: Unifying Object Detection Heads with Attentions","date":"2021-06-15","arxiv_id":"2106.08322","rows_on_this_dataset":9,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":8,"samples_constructed":5,"samples_ran_checked":6,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["microsoft/DynamicHead"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dynamic-head-unifying-object-detection-heads#ran","syntology_url":"https://syntology.ai/paper/2106.08322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08322"}}}},{"paper":"/paper/mltr-multi-label-classification-with","slug":"mltr-multi-label-classification-with","title":"MlTr: Multi-label Classification with Transformer","date":"2021-06-11","arxiv_id":"2106.06195","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/detreg-unsupervised-pretraining-with-region","slug":"detreg-unsupervised-pretraining-with-region","title":"DETReg: Unsupervised Pretraining with Region Priors for Object Detection","date":"2021-06-08","arxiv_id":"2106.04550","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":1,"official":{"repos":["amirbar/detreg"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/detreg-unsupervised-pretraining-with-region#ran","syntology_url":"https://syntology.ai/paper/2106.04550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04550"}}}},{"paper":"/paper/combinatorial-optimization-for-panoptic","slug":"combinatorial-optimization-for-panoptic","title":"Combinatorial Optimization for Panoptic Segmentation: A Fully Differentiable Approach","date":"2021-06-06","arxiv_id":"2106.03188","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/solq-segmenting-objects-by-learning-queries","slug":"solq-segmenting-objects-by-learning-queries","title":"SOLQ: Segmenting Objects by Learning Queries","date":"2021-06-04","arxiv_id":"2106.02351","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/x-volution-on-the-unification-of-convolution","slug":"x-volution-on-the-unification-of-convolution","title":"X-volution: On the unification of convolution and self-attention","date":"2021-06-04","arxiv_id":"2106.02253","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/learning-relation-alignment-for-calibrated","slug":"learning-relation-alignment-for-calibrated","title":"Learning Relation Alignment for Calibrated Cross-modal Retrieval","date":"2021-05-28","arxiv_id":"2105.13868","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["lancopku/IAIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/learning-relation-alignment-for-calibrated#ran","syntology_url":"https://syntology.ai/paper/2105.13868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13868"}}}},{"paper":"/paper/cogview-mastering-text-to-image-generation","slug":"cogview-mastering-text-to-image-generation","title":"CogView: Mastering Text-to-Image Generation via Transformers","date":"2021-05-26","arxiv_id":"2105.13290","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["THUDM/CogView"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogview-mastering-text-to-image-generation#ran","syntology_url":"https://syntology.ai/paper/2105.13290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.13290"}}}},{"paper":"/paper/vipnas-efficient-video-pose-estimation-via","slug":"vipnas-efficient-video-pose-estimation-via","title":"ViPNAS: Efficient Video Pose Estimation via Neural Architecture Search","date":"2021-05-21","arxiv_id":"2105.10154","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["luminxu/ViPNAS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vipnas-efficient-video-pose-estimation-via#ran","syntology_url":"https://syntology.ai/paper/2105.10154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.10154"}}}},{"paper":"/paper/discobox-weakly-supervised-instance","slug":"discobox-weakly-supervised-instance","title":"DiscoBox: Weakly Supervised Instance Segmentation and Semantic Correspondence from Box Supervision","date":"2021-05-13","arxiv_id":"2105.06464","rows_on_this_dataset":4,"code_links":3,"syntology":null},{"paper":"/paper/you-only-learn-one-representation-unified","slug":"you-only-learn-one-representation-unified","title":"You Only Learn One Representation: Unified Network for Multiple Tasks","date":"2021-05-10","arxiv_id":"2105.04206","rows_on_this_dataset":8,"code_links":9,"syntology":null},{"paper":"/paper/polarmask-enhanced-polar-representation-for","slug":"polarmask-enhanced-polar-representation-for","title":"PolarMask++: Enhanced Polar Representation for Single-Shot Instance Segmentation and Beyond","date":"2021-05-05","arxiv_id":"2105.02184","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/queryinst-parallelly-supervised-mask-query","slug":"queryinst-parallelly-supervised-mask-query","title":"Instances as Queries","date":"2021-05-05","arxiv_id":"2105.01928","rows_on_this_dataset":4,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["hustvl/QueryInst"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/queryinst-parallelly-supervised-mask-query#ran","syntology_url":"https://syntology.ai/paper/2105.01928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.01928"}}}},{"paper":"/paper/istr-end-to-end-instance-segmentation-with","slug":"istr-end-to-end-instance-segmentation-with","title":"ISTR: End-to-End Instance Segmentation with Transformers","date":"2021-05-03","arxiv_id":"2105.00637","rows_on_this_dataset":6,"code_links":1,"syntology":null},{"paper":"/paper/imagenet-21k-pretraining-for-the-masses","slug":"imagenet-21k-pretraining-for-the-masses","title":"ImageNet-21K Pretraining for the Masses","date":"2021-04-22","arxiv_id":"2104.10972","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["Alibaba-MIIL/ImageNet21K"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/imagenet-21k-pretraining-for-the-masses#ran","syntology_url":"https://syntology.ai/paper/2104.10972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.10972"}}}},{"paper":"/paper/lite-hrnet-a-lightweight-high-resolution","slug":"lite-hrnet-a-lightweight-high-resolution","title":"Lite-HRNet: A Lightweight High-Resolution Network","date":"2021-04-13","arxiv_id":"2104.06403","rows_on_this_dataset":2,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":29,"samples_ran":13,"samples_constructed":2,"samples_ran_checked":11,"samples_ran_instrument_failed":2,"samples_unverified":16,"pointer_only_for_licence":13,"official":{"repos":["HRNet/Lite-HRNet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lite-hrnet-a-lightweight-high-resolution#ran","syntology_url":"https://syntology.ai/paper/2104.06403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06403"}}}},{"paper":"/paper/location-sensitive-visual-recognition-with","slug":"location-sensitive-visual-recognition-with","title":"Location-Sensitive Visual Recognition with Cross-IOU Loss","date":"2021-04-11","arxiv_id":"2104.04899","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/multiple-instance-active-learning-for-object","slug":"multiple-instance-active-learning-for-object","title":"Multiple instance active learning for object detection","date":"2021-04-06","arxiv_id":"2104.02324","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["yuantn/MI-AOD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/multiple-instance-active-learning-for-object#ran","syntology_url":"https://syntology.ai/paper/2104.02324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.02324"}}}},{"paper":"/paper/weakly-supervised-instance-segmentation-via","slug":"weakly-supervised-instance-segmentation-via","title":"Weakly-supervised Instance Segmentation via Class-agnostic Learning with Salient Images","date":"2021-04-04","arxiv_id":"2104.01526","rows_on_this_dataset":1,"code_links":0,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/weakly-supervised-instance-segmentation-via#ran","syntology_url":"https://syntology.ai/paper/2104.01526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.01526"}}}}],"record_sha256":"cb7578bbdb97233e243cdf637282230644f6000effabb2f974766b6bee5b4990","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}