{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/coco/papers/ran/1","list_of":"/dataset/coco","dataset":"COCO (Common Objects in Context)","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this dataset or check it against this dataset's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":256,"counts":{"papers_with_a_benchmark_row":579,"with_a_code_link":504,"where_syntology_ran_a_sample":256,"not_listed_spam_title":0,"listed":579,"listed_where_code_ran":256,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":226,"every_run_a_failure_of_syntologys_instrument":30,"listed_with_a_run_with_no_instrument_failure":226,"listed_every_run_a_failure_of_syntologys_instrument":30,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/coco/papers/ran/1","prev":null,"next":"/dataset/coco/papers/ran/2","papers":[{"paper":"/paper/decoupling-classifier-for-boosting-few-shot-1","slug":"decoupling-classifier-for-boosting-few-shot-1","title":"Decoupling Classifier for Boosting Few-shot Object Detection and Instance Segmentation","date":"2025-05-20","arxiv_id":"2505.14239","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/decoupling-classifier-for-boosting-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2505.14239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14239"}}}},{"paper":"/paper/perception-encoder-the-best-visual-embeddings","slug":"perception-encoder-the-best-visual-embeddings","title":"Perception Encoder: The best visual embeddings are not at the output of the network","date":"2025-04-17","arxiv_id":"2504.13181","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":14,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":5,"samples_unverified":4,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/perception-encoder-the-best-visual-embeddings#ran","syntology_url":"https://syntology.ai/paper/2504.13181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13181"}}}},{"paper":"/paper/your-vit-is-secretly-an-image-segmentation-1","slug":"your-vit-is-secretly-an-image-segmentation-1","title":"Your ViT is Secretly an Image Segmentation Model","date":"2025-03-24","arxiv_id":"2503.19108","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["tue-mps/eomt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/your-vit-is-secretly-an-image-segmentation-1#ran","syntology_url":"https://syntology.ai/paper/2503.19108","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19108"}}}},{"paper":"/paper/deim-detr-with-improved-matching-for-fast","slug":"deim-detr-with-improved-matching-for-fast","title":"DEIM: DETR with Improved Matching for Fast Convergence","date":"2024-12-05","arxiv_id":"2412.04234","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":13,"official":{"repos":["shihuahuang95/deim"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deim-detr-with-improved-matching-for-fast#ran","syntology_url":"https://syntology.ai/paper/2412.04234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04234"}}}},{"paper":"/paper/hyperseg-towards-universal-visual","slug":"hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","arxiv_id":"2411.17606","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":4,"samples_unverified":4,"pointer_only_for_licence":2,"official":{"repos":["congvvc/HyperSeg"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hyperseg-towards-universal-visual#ran","syntology_url":"https://syntology.ai/paper/2411.17606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17606"}}}},{"paper":"/paper/d-fine-redefine-regression-task-in-detrs-as","slug":"d-fine-redefine-regression-task-in-detrs-as","title":"D-FINE: Redefine Regression Task in DETRs as Fine-grained Distribution Refinement","date":"2024-10-17","arxiv_id":"2410.13842","rows_on_this_dataset":7,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":0,"official":{"repos":["Peterande/D-FINE"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/d-fine-redefine-regression-task-in-detrs-as#ran","syntology_url":"https://syntology.ai/paper/2410.13842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13842"}}}},{"paper":"/paper/sapiens-foundation-for-human-vision-models","slug":"sapiens-foundation-for-human-vision-models","title":"Sapiens: Foundation for Human Vision Models","date":"2024-08-22","arxiv_id":"2408.12569","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sapiens-foundation-for-human-vision-models#ran","syntology_url":"https://syntology.ai/paper/2408.12569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12569"}}}},{"paper":"/paper/relation-detr-exploring-explicit-position","slug":"relation-detr-exploring-explicit-position","title":"Relation DETR: Exploring Explicit Position Relation Prior for Object Detection","date":"2024-07-16","arxiv_id":"2407.11699","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["xiuqhou/relation-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/relation-detr-exploring-explicit-position#ran","syntology_url":"https://syntology.ai/paper/2407.11699","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11699"}}}},{"paper":"/paper/long-and-short-guidance-in-score-identity","slug":"long-and-short-guidance-in-score-identity","title":"Long and Short Guidance in Score identity Distillation for One-Step Text-to-Image Generation","date":"2024-06-03","arxiv_id":"2406.01561","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":24,"samples_ran":19,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":6,"samples_unverified":5,"pointer_only_for_licence":7,"official":{"repos":["mingyuanzhou/sid-lsg"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["community","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/long-and-short-guidance-in-score-identity#ran","syntology_url":"https://syntology.ai/paper/2406.01561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01561"}}}},{"paper":"/paper/yolov10-real-time-end-to-end-object-detection","slug":"yolov10-real-time-end-to-end-object-detection","title":"YOLOv10: Real-Time End-to-End Object Detection","date":"2024-05-23","arxiv_id":"2405.14458","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["THU-MIG/yolov10"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov10-real-time-end-to-end-object-detection#ran","syntology_url":"https://syntology.ai/paper/2405.14458","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14458"}}}},{"paper":"/paper/isearle-improving-textual-inversion-for-zero","slug":"isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":4,"official":{"repos":["miccunifi/circo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/isearle-improving-textual-inversion-for-zero#ran","syntology_url":"https://syntology.ai/paper/2405.02951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02951"}}}},{"paper":"/paper/3shnet-boosting-image-sentence-retrieval-via","slug":"3shnet-boosting-image-sentence-retrieval-via","title":"3SHNet: Boosting Image-Sentence Retrieval via Visual Semantic-Spatial Self-Highlighting","date":"2024-04-26","arxiv_id":"2404.17273","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["xurige1995/3shnet"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/3shnet-boosting-image-sentence-retrieval-via#ran","syntology_url":"https://syntology.ai/paper/2404.17273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.17273"}}}},{"paper":"/paper/yolov9-learning-what-you-want-to-learn-using","slug":"yolov9-learning-what-you-want-to-learn-using","title":"YOLOv9: Learning What You Want to Learn Using Programmable Gradient Information","date":"2024-02-21","arxiv_id":"2402.13616","rows_on_this_dataset":8,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":32,"samples_ran":25,"samples_constructed":9,"samples_ran_checked":24,"samples_ran_instrument_failed":1,"samples_unverified":7,"pointer_only_for_licence":11,"official":{"repos":["WongKinYiu/YOLO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["listed","named_in_paper","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov9-learning-what-you-want-to-learn-using#ran","syntology_url":"https://syntology.ai/paper/2402.13616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13616"}}}},{"paper":"/paper/complete-instances-mining-for-weakly-1","slug":"complete-instances-mining-for-weakly-1","title":"Complete Instances Mining for Weakly Supervised Instance Segmentation","date":"2024-02-12","arxiv_id":"2402.07633","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["ZechengLi19/CIM"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/complete-instances-mining-for-weakly-1#ran","syntology_url":"https://syntology.ai/paper/2402.07633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07633"}}}},{"paper":"/paper/cross-domain-few-shot-object-detection-via","slug":"cross-domain-few-shot-object-detection-via","title":"Cross-Domain Few-Shot Object Detection via Enhanced Open-Set Object Detector","date":"2024-02-05","arxiv_id":"2402.03094","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["lovelyqian/CDFSOD-benchmark"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cross-domain-few-shot-object-detection-via#ran","syntology_url":"https://syntology.ai/paper/2402.03094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03094"}}}},{"paper":"/paper/harnessing-diffusion-models-for-visual","slug":"harnessing-diffusion-models-for-visual","title":"Harnessing Diffusion Models for Visual Perception with Meta Prompts","date":"2023-12-22","arxiv_id":"2312.14733","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":4,"official":{"repos":["fudan-zvg/meta-prompts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/harnessing-diffusion-models-for-visual#ran","syntology_url":"https://syntology.ai/paper/2312.14733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14733"}}}},{"paper":"/paper/internvl-scaling-up-vision-foundation-models","slug":"internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["opengvlab/internvl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/internvl-scaling-up-vision-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2312.14238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14238"}}}},{"paper":"/paper/general-object-foundation-model-for-images","slug":"general-object-foundation-model-for-images","title":"General Object Foundation Model for Images and Videos at Scale","date":"2023-12-14","arxiv_id":"2312.09158","rows_on_this_dataset":12,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["FoundationVision/GLEE"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/general-object-foundation-model-for-images#ran","syntology_url":"https://syntology.ai/paper/2312.09158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09158"}}}},{"paper":"/paper/hulk-a-universal-knowledge-translator-for","slug":"hulk-a-universal-knowledge-translator-for","title":"Hulk: A Universal Knowledge Translator for Human-Centric Tasks","date":"2023-12-04","arxiv_id":"2312.01697","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":24,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":0,"samples_unverified":11,"pointer_only_for_licence":11,"official":{"repos":["opengvlab/hulk","opengvlab/humanbench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":11,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hulk-a-universal-knowledge-translator-for#ran","syntology_url":"https://syntology.ai/paper/2312.01697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01697"}}}},{"paper":"/paper/transnext-robust-foveal-visual-perception-for","slug":"transnext-robust-foveal-visual-perception-for","title":"TransNeXt: Robust Foveal Visual Perception for Vision Transformers","date":"2023-11-28","arxiv_id":"2311.17132","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["daishiresearch/transnext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/transnext-robust-foveal-visual-perception-for#ran","syntology_url":"https://syntology.ai/paper/2311.17132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17132"}}}},{"paper":"/paper/unireplknet-a-universal-perception-large","slug":"unireplknet-a-universal-perception-large","title":"UniRepLKNet: A Universal Perception Large-Kernel ConvNet for Audio, Video, Point Cloud, Time-Series and Image Recognition","date":"2023-11-27","arxiv_id":"2311.15599","rows_on_this_dataset":6,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["ailab-cvc/unireplknet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/unireplknet-a-universal-perception-large#ran","syntology_url":"https://syntology.ai/paper/2311.15599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15599"}}}},{"paper":"/paper/unipose-detecting-any-keypoints","slug":"unipose-detecting-any-keypoints","title":"X-Pose: Detecting Any Keypoints","date":"2023-10-12","arxiv_id":"2310.08530","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":13,"official":{"repos":["IDEA-Research/UniPose","idea-research/x-pose"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/unipose-detecting-any-keypoints#ran","syntology_url":"https://syntology.ai/paper/2310.08530","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08530"}}}},{"paper":"/paper/kandinsky-an-improved-text-to-image-synthesis","slug":"kandinsky-an-improved-text-to-image-synthesis","title":"Kandinsky: an Improved Text-to-Image Synthesis with Image Prior and Latent Diffusion","date":"2023-10-05","arxiv_id":"2310.03502","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":12,"samples_constructed":1,"samples_ran_checked":7,"samples_ran_instrument_failed":5,"samples_unverified":6,"pointer_only_for_licence":10,"official":{"repos":["ai-forever/Kandinsky-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/kandinsky-an-improved-text-to-image-synthesis#ran","syntology_url":"https://syntology.ai/paper/2310.03502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03502"}}}},{"paper":"/paper/context-i2w-mapping-images-to-context","slug":"context-i2w-mapping-images-to-context","title":"Context-I2W: Mapping Images to Context-dependent Words for Accurate Zero-Shot Composed Image Retrieval","date":"2023-09-28","arxiv_id":"2309.16137","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["pter61/context-i2w"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/context-i2w-mapping-images-to-context#ran","syntology_url":"https://syntology.ai/paper/2309.16137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16137"}}}},{"paper":"/paper/mocae-mixture-of-calibrated-experts","slug":"mocae-mixture-of-calibrated-experts","title":"MoCaE: Mixture of Calibrated Experts Significantly Improves Object Detection","date":"2023-09-26","arxiv_id":"2309.14976","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["fiveai/MoCaE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mocae-mixture-of-calibrated-experts#ran","syntology_url":"https://syntology.ai/paper/2309.14976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14976"}}}},{"paper":"/paper/detect-every-thing-with-few-examples","slug":"detect-every-thing-with-few-examples","title":"Detect Everything with Few Examples","date":"2023-09-22","arxiv_id":"2309.12969","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":17,"samples_constructed":0,"samples_ran_checked":17,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":1,"official":{"repos":["mlzxy/devit"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/detect-every-thing-with-few-examples#ran","syntology_url":"https://syntology.ai/paper/2309.12969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.12969"}}}},{"paper":"/paper/dat-spatially-dynamic-vision-transformer-with","slug":"dat-spatially-dynamic-vision-transformer-with","title":"DAT++: Spatially Dynamic Vision Transformer with Deformable Attention","date":"2023-09-04","arxiv_id":"2309.01430","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":6,"official":{"repos":["leaplabthu/dat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dat-spatially-dynamic-vision-transformer-with#ran","syntology_url":"https://syntology.ai/paper/2309.01430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01430"}}}},{"paper":"/paper/hierarchical-open-vocabulary-universal-image-1","slug":"hierarchical-open-vocabulary-universal-image-1","title":"Hierarchical Open-vocabulary Universal Image Segmentation","date":"2023-07-03","arxiv_id":"2307.00764","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["berkeley-hipie/hipie"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hierarchical-open-vocabulary-universal-image-1#ran","syntology_url":"https://syntology.ai/paper/2307.00764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.00764"}}}},{"paper":"/paper/rethinking-pose-estimation-in-crowds","slug":"rethinking-pose-estimation-in-crowds","title":"Rethinking pose estimation in crowds: overcoming the detection information-bottleneck and ambiguity","date":"2023-06-13","arxiv_id":"2306.07879","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":3,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["amathislab/BUCTD"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rethinking-pose-estimation-in-crowds#ran","syntology_url":"https://syntology.ai/paper/2306.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07879"}}}},{"paper":"/paper/hiera-a-hierarchical-vision-transformer","slug":"hiera-a-hierarchical-vision-transformer","title":"Hiera: A Hierarchical Vision Transformer without the Bells-and-Whistles","date":"2023-06-01","arxiv_id":"2306.00989","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["facebookresearch/hiera"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hiera-a-hierarchical-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2306.00989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.00989"}}}},{"paper":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":42,"samples_ran":35,"samples_constructed":4,"samples_ran_checked":29,"samples_ran_instrument_failed":6,"samples_unverified":7,"pointer_only_for_licence":8,"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}}}},{"paper":"/paper/one-peace-exploring-one-general","slug":"one-peace-exploring-one-general","title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities","date":"2023-05-18","arxiv_id":"2305.11172","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":2,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":1,"official":{"repos":["OFA-Sys/ONE-PEACE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/one-peace-exploring-one-general#ran","syntology_url":"https://syntology.ai/paper/2305.11172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11172"}}}},{"paper":"/paper/region-aware-pretraining-for-open-vocabulary","slug":"region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","arxiv_id":"2305.07011","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":5,"samples_constructed":3,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["mcahny/rovit","google-research/google-research"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/region-aware-pretraining-for-open-vocabulary#ran","syntology_url":"https://syntology.ai/paper/2305.07011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07011"}}}},{"paper":"/paper/tr0n-translator-networks-for-0-shot-plug-and","slug":"tr0n-translator-networks-for-0-shot-plug-and","title":"TR0N: Translator Networks for 0-Shot Plug-and-Play Conditional Generation","date":"2023-04-26","arxiv_id":"2304.13742","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":19,"samples_ran":17,"samples_constructed":1,"samples_ran_checked":12,"samples_ran_instrument_failed":5,"samples_unverified":2,"pointer_only_for_licence":9,"official":{"repos":["layer6ai-labs/tr0n"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tr0n-translator-networks-for-0-shot-plug-and#ran","syntology_url":"https://syntology.ai/paper/2304.13742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13742"}}}},{"paper":"/paper/detrs-beat-yolos-on-real-time-object","slug":"detrs-beat-yolos-on-real-time-object","title":"DETRs Beat YOLOs on Real-time Object Detection","date":"2023-04-17","arxiv_id":"2304.08069","rows_on_this_dataset":4,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["lyuwenyu/RT-DETR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/detrs-beat-yolos-on-real-time-object#ran","syntology_url":"https://syntology.ai/paper/2304.08069","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08069"}}}},{"paper":"/paper/layoutdiffusion-controllable-diffusion-model","slug":"layoutdiffusion-controllable-diffusion-model","title":"LayoutDiffusion: Controllable Diffusion Model for Layout-to-image Generation","date":"2023-03-30","arxiv_id":"2303.17189","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":19,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":1,"samples_unverified":8,"pointer_only_for_licence":3,"official":{"repos":["dcdcvgroup/layout-diffusion-mindspore"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["named_in_paper","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/layoutdiffusion-controllable-diffusion-model#ran","syntology_url":"https://syntology.ai/paper/2303.17189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17189"}}}},{"paper":"/paper/beyond-appearance-a-semantic-controllable","slug":"beyond-appearance-a-semantic-controllable","title":"Beyond Appearance: a Semantic Controllable Self-Supervised Learning Framework for Human-Centric Visual Tasks","date":"2023-03-30","arxiv_id":"2303.17602","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["tinyvision/SOLIDER"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/beyond-appearance-a-semantic-controllable#ran","syntology_url":"https://syntology.ai/paper/2303.17602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17602"}}}},{"paper":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}}}},{"paper":"/paper/human-pose-as-compositional-tokens","slug":"human-pose-as-compositional-tokens","title":"Human Pose as Compositional Tokens","date":"2023-03-21","arxiv_id":"2303.11638","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["gengzigang/pct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/human-pose-as-compositional-tokens#ran","syntology_url":"https://syntology.ai/paper/2303.11638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11638"}}}},{"paper":"/paper/biformer-vision-transformer-with-bi-level","slug":"biformer-vision-transformer-with-bi-level","title":"BiFormer: Vision Transformer with Bi-Level Routing Attention","date":"2023-03-15","arxiv_id":"2303.08810","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":5,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["rayleizhu/biformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/biformer-vision-transformer-with-bi-level#ran","syntology_url":"https://syntology.ai/paper/2303.08810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08810"}}}},{"paper":"/paper/universal-instance-perception-as-object","slug":"universal-instance-perception-as-object","title":"Universal Instance Perception as Object Discovery and Retrieval","date":"2023-03-12","arxiv_id":"2303.06674","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["MasterBin-IIAU/UNINEXT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/universal-instance-perception-as-object#ran","syntology_url":"https://syntology.ai/paper/2303.06674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06674"}}}},{"paper":"/paper/humanbench-towards-general-human-centric","slug":"humanbench-towards-general-human-centric","title":"HumanBench: Towards General Human-centric Perception with Projector Assisted Pretraining","date":"2023-03-10","arxiv_id":"2303.05675","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":25,"samples_ran":15,"samples_constructed":1,"samples_ran_checked":8,"samples_ran_instrument_failed":7,"samples_unverified":10,"pointer_only_for_licence":0,"official":{"repos":["opengvlab/humanbench"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":10,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/humanbench-towards-general-human-centric#ran","syntology_url":"https://syntology.ai/paper/2303.05675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05675"}}}},{"paper":"/paper/grounding-dino-marrying-dino-with-grounded","slug":"grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","arxiv_id":"2303.05499","rows_on_this_dataset":2,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":2,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["idea-research/groundingdino"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/grounding-dino-marrying-dino-with-grounded#ran","syntology_url":"https://syntology.ai/paper/2303.05499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05499"}}}},{"paper":"/paper/scaling-up-gans-for-text-to-image-synthesis","slug":"scaling-up-gans-for-text-to-image-synthesis","title":"Scaling up GANs for Text-to-Image Synthesis","date":"2023-03-09","arxiv_id":"2303.05511","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":3,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/scaling-up-gans-for-text-to-image-synthesis#ran","syntology_url":"https://syntology.ai/paper/2303.05511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05511"}}}},{"paper":"/paper/unihcp-a-unified-model-for-human-centric","slug":"unihcp-a-unified-model-for-human-centric","title":"UniHCP: A Unified Model for Human-Centric Perceptions","date":"2023-03-06","arxiv_id":"2303.02936","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":5,"pointer_only_for_licence":10,"official":{"repos":["opengvlab/unihcp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/unihcp-a-unified-model-for-human-centric#ran","syntology_url":"https://syntology.ai/paper/2303.02936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02936"}}}},{"paper":"/paper/pic2word-mapping-pictures-to-words-for-zero","slug":"pic2word-mapping-pictures-to-words-for-zero","title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","date":"2023-02-06","arxiv_id":"2302.03084","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["google-research/composed_image_retrieval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/pic2word-mapping-pictures-to-words-for-zero#ran","syntology_url":"https://syntology.ai/paper/2302.03084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.03084"}}}},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","rows_on_this_dataset":4,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":4,"samples_constructed":4,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":1,"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}}}},{"paper":"/paper/galip-generative-adversarial-clips-for-text","slug":"galip-generative-adversarial-clips-for-text","title":"GALIP: Generative Adversarial CLIPs for Text-to-Image Synthesis","date":"2023-01-30","arxiv_id":"2301.12959","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":1,"samples_unverified":4,"pointer_only_for_licence":2,"official":{"repos":["tobran/galip"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/galip-generative-adversarial-clips-for-text#ran","syntology_url":"https://syntology.ai/paper/2301.12959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12959"}}}},{"paper":"/paper/gligen-open-set-grounded-text-to-image","slug":"gligen-open-set-grounded-text-to-image","title":"GLIGEN: Open-Set Grounded Text-to-Image Generation","date":"2023-01-17","arxiv_id":"2301.07093","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["gligen/GLIGEN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/gligen-open-set-grounded-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2301.07093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07093"}}}},{"paper":"/paper/yolov6-v3-0-a-full-scale-reloading","slug":"yolov6-v3-0-a-full-scale-reloading","title":"YOLOv6 v3.0: A Full-Scale Reloading","date":"2023-01-13","arxiv_id":"2301.05586","rows_on_this_dataset":6,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["meituan/yolov6"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov6-v3-0-a-full-scale-reloading#ran","syntology_url":"https://syntology.ai/paper/2301.05586","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.05586"}}}},{"paper":"/paper/muse-text-to-image-generation-via-masked","slug":"muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","arxiv_id":"2301.00704","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":21,"samples_ran":19,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":11,"samples_unverified":2,"pointer_only_for_licence":11,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/muse-text-to-image-generation-via-masked#ran","syntology_url":"https://syntology.ai/paper/2301.00704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00704"}}}},{"paper":"/paper/reversible-column-networks","slug":"reversible-column-networks","title":"Reversible Column Networks","date":"2022-12-22","arxiv_id":"2212.11696","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":8,"samples_constructed":8,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["megvii-research/revcol"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/reversible-column-networks#ran","syntology_url":"https://syntology.ai/paper/2212.11696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.11696"}}}},{"paper":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}}}},{"paper":"/paper/rtmdet-an-empirical-study-of-designing-real","slug":"rtmdet-an-empirical-study-of-designing-real","title":"RTMDet: An Empirical Study of Designing Real-Time Object Detectors","date":"2022-12-14","arxiv_id":"2212.07784","rows_on_this_dataset":2,"code_links":14,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":16,"samples_constructed":0,"samples_ran_checked":16,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":3,"official":{"repos":["open-mmlab/mmdetection"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rtmdet-an-empirical-study-of-designing-real#ran","syntology_url":"https://syntology.ai/paper/2212.07784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.07784"}}}},{"paper":"/paper/deepcut-unsupervised-segmentation-using-graph","slug":"deepcut-unsupervised-segmentation-using-graph","title":"DeepCut: Unsupervised Segmentation using Graph Neural Networks Clustering","date":"2022-12-12","arxiv_id":"2212.05853","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deepcut-unsupervised-segmentation-using-graph#ran","syntology_url":"https://syntology.ai/paper/2212.05853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.05853"}}}},{"paper":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip","slug":"x-paste-revisit-copy-paste-at-scale-with-clip","title":"X-Paste: Revisiting Scalable Copy-Paste for Instance Segmentation using CLIP and StableDiffusion","date":"2022-12-07","arxiv_id":"2212.03863","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["yoctta/xpaste"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip#ran","syntology_url":"https://syntology.ai/paper/2212.03863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.03863"}}}},{"paper":"/paper/diffusioninst-diffusion-model-for-instance","slug":"diffusioninst-diffusion-model-for-instance","title":"DiffusionInst: Diffusion Model for Instance Segmentation","date":"2022-12-06","arxiv_id":"2212.02773","rows_on_this_dataset":4,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":16,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":5,"samples_unverified":4,"pointer_only_for_licence":7,"official":{"repos":["chenhaoxing/DiffusionInst"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/diffusioninst-diffusion-model-for-instance#ran","syntology_url":"https://syntology.ai/paper/2212.02773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.02773"}}}},{"paper":"/paper/grit-a-generative-region-to-text-transformer","slug":"grit-a-generative-region-to-text-transformer","title":"GRiT: A Generative Region-to-text Transformer for Object Understanding","date":"2022-12-01","arxiv_id":"2212.00280","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":8,"official":{"repos":["JialianW/GRiT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/grit-a-generative-region-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2212.00280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00280"}}}},{"paper":"/paper/shifted-diffusion-for-text-to-image","slug":"shifted-diffusion-for-text-to-image","title":"Shifted Diffusion for Text-to-image Generation","date":"2022-11-24","arxiv_id":"2211.15388","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":3,"samples_unverified":4,"pointer_only_for_licence":6,"official":{"repos":["drboog/Shifted_Diffusion"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/shifted-diffusion-for-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2211.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15388"}}}},{"paper":"/paper/damo-yolo-a-report-on-real-time-object","slug":"damo-yolo-a-report-on-real-time-object","title":"DAMO-YOLO : A Report on Real-Time Object Detection Design","date":"2022-11-23","arxiv_id":"2211.15444","rows_on_this_dataset":4,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":7,"pointer_only_for_licence":12,"official":{"repos":["alibaba/lightweight-neural-architecture-search","tinyvision/damo-yolo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/damo-yolo-a-report-on-real-time-object#ran","syntology_url":"https://syntology.ai/paper/2211.15444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.15444"}}}},{"paper":"/paper/x-2-vlm-all-in-one-pre-trained-model-for","slug":"x-2-vlm-all-in-one-pre-trained-model-for","title":"X$^2$-VLM: All-In-One Pre-trained Model For Vision-Language Tasks","date":"2022-11-22","arxiv_id":"2211.12402","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":6,"official":{"repos":["zengyan-97/x2-vlm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/x-2-vlm-all-in-one-pre-trained-model-for#ran","syntology_url":"https://syntology.ai/paper/2211.12402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12402"}}}},{"paper":"/paper/detrs-with-collaborative-hybrid-assignments","slug":"detrs-with-collaborative-hybrid-assignments","title":"DETRs with Collaborative Hybrid Assignments Training","date":"2022-11-22","arxiv_id":"2211.12860","rows_on_this_dataset":6,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/detrs-with-collaborative-hybrid-assignments#ran","syntology_url":"https://syntology.ai/paper/2211.12860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.12860"}}}},{"paper":"/paper/eva-exploring-the-limits-of-masked-visual","slug":"eva-exploring-the-limits-of-masked-visual","title":"EVA: Exploring the Limits of Masked Visual Representation Learning at Scale","date":"2022-11-14","arxiv_id":"2211.07636","rows_on_this_dataset":4,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["baaivision/eva","rwightman/pytorch-image-models"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/eva-exploring-the-limits-of-masked-visual#ran","syntology_url":"https://syntology.ai/paper/2211.07636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.07636"}}}},{"paper":"/paper/internimage-exploring-large-scale-vision","slug":"internimage-exploring-large-scale-vision","title":"InternImage: Exploring Large-Scale Vision Foundation Models with Deformable Convolutions","date":"2022-11-10","arxiv_id":"2211.05778","rows_on_this_dataset":11,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["opengvlab/internimage"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/internimage-exploring-large-scale-vision#ran","syntology_url":"https://syntology.ai/paper/2211.05778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.05778"}}}},{"paper":"/paper/oneformer-one-transformer-to-rule-universal","slug":"oneformer-one-transformer-to-rule-universal","title":"OneFormer: One Transformer to Rule Universal Image Segmentation","date":"2022-11-10","arxiv_id":"2211.06220","rows_on_this_dataset":6,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["SHI-Labs/OneFormer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/oneformer-one-transformer-to-rule-universal#ran","syntology_url":"https://syntology.ai/paper/2211.06220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.06220"}}}},{"paper":"/paper/efficient-multi-order-gated-aggregation","slug":"efficient-multi-order-gated-aggregation","title":"MogaNet: Multi-order Gated Aggregation Network","date":"2022-11-07","arxiv_id":"2211.03295","rows_on_this_dataset":10,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":12,"samples_constructed":7,"samples_ran_checked":10,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["Westlake-AI/MogaNet","Westlake-AI/openmixup","chengtan9907/OpenSTL"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":7,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficient-multi-order-gated-aggregation#ran","syntology_url":"https://syntology.ai/paper/2211.03295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.03295"}}}},{"paper":"/paper/ediffi-text-to-image-diffusion-models-with-an","slug":"ediffi-text-to-image-diffusion-models-with-an","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","date":"2022-11-02","arxiv_id":"2211.01324","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ediffi-text-to-image-diffusion-models-with-an#ran","syntology_url":"https://syntology.ai/paper/2211.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.01324"}}}},{"paper":"/paper/towards-sustainable-self-supervised-learning","slug":"towards-sustainable-self-supervised-learning","title":"Towards Sustainable Self-supervised Learning","date":"2022-10-20","arxiv_id":"2210.11016","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":6,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":8,"official":{"repos":["sail-sg/tec"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/towards-sustainable-self-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2210.11016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.11016"}}}},{"paper":"/paper/move-unsupervised-movable-object-segmentation","slug":"move-unsupervised-movable-object-segmentation","title":"MOVE: Unsupervised Movable Object Segmentation and Detection","date":"2022-10-14","arxiv_id":"2210.07920","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":1,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":6,"official":{"repos":["adambielski/move-seg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/move-unsupervised-movable-object-segmentation#ran","syntology_url":"https://syntology.ai/paper/2210.07920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07920"}}}},{"paper":"/paper/all-are-worth-words-a-vit-backbone-for-score","slug":"all-are-worth-words-a-vit-backbone-for-score","title":"All are Worth Words: A ViT Backbone for Diffusion Models","date":"2022-09-25","arxiv_id":"2209.12152","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["baofff/U-ViT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/all-are-worth-words-a-vit-backbone-for-score#ran","syntology_url":"https://syntology.ai/paper/2209.12152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.12152"}}}},{"paper":"/paper/exploring-target-representations-for-masked","slug":"exploring-target-representations-for-masked","title":"Exploring Target Representations for Masked Autoencoders","date":"2022-09-08","arxiv_id":"2209.03917","rows_on_this_dataset":8,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["liuxingbin/dbot"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/exploring-target-representations-for-masked#ran","syntology_url":"https://syntology.ai/paper/2209.03917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03917"}}}},{"paper":"/paper/yolov6-a-single-stage-object-detection","slug":"yolov6-a-single-stage-object-detection","title":"YOLOv6: A Single-Stage Object Detection Framework for Industrial Applications","date":"2022-09-07","arxiv_id":"2209.02976","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["meituan/yolov6"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov6-a-single-stage-object-detection#ran","syntology_url":"https://syntology.ai/paper/2209.02976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.02976"}}}},{"paper":"/paper/analog-bits-generating-discrete-data-using","slug":"analog-bits-generating-discrete-data-using","title":"Analog Bits: Generating Discrete Data using Diffusion Models with Self-Conditioning","date":"2022-08-08","arxiv_id":"2208.04202","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":20,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":9,"samples_unverified":7,"pointer_only_for_licence":4,"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/analog-bits-generating-discrete-data-using#ran","syntology_url":"https://syntology.ai/paper/2208.04202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.04202"}}}},{"paper":"/paper/hornet-efficient-high-order-spatial","slug":"hornet-efficient-high-order-spatial","title":"HorNet: Efficient High-Order Spatial Interactions with Recursive Gated Convolutions","date":"2022-07-28","arxiv_id":"2207.14284","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["raoyongming/hornet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hornet-efficient-high-order-spatial#ran","syntology_url":"https://syntology.ai/paper/2207.14284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.14284"}}}},{"paper":"/paper/yolov7-trainable-bag-of-freebies-sets-new","slug":"yolov7-trainable-bag-of-freebies-sets-new","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","date":"2022-07-06","arxiv_id":"2207.02696","rows_on_this_dataset":10,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":5,"official":{"repos":["wongkinyiu/yolov7"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov7-trainable-bag-of-freebies-sets-new#ran","syntology_url":"https://syntology.ai/paper/2207.02696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.02696"}}}},{"paper":"/paper/cmt-deeplab-clustering-mask-transformers-for-1","slug":"cmt-deeplab-clustering-mask-transformers-for-1","title":"CMT-DeepLab: Clustering Mask Transformers for Panoptic Segmentation","date":"2022-06-17","arxiv_id":"2206.08948","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cmt-deeplab-clustering-mask-transformers-for-1#ran","syntology_url":"https://syntology.ai/paper/2206.08948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08948"}}}},{"paper":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}}}},{"paper":"/paper/mask-dino-towards-a-unified-transformer-based-1","slug":"mask-dino-towards-a-unified-transformer-based-1","title":"Mask DINO: Towards A Unified Transformer-based Framework for Object Detection and Segmentation","date":"2022-06-06","arxiv_id":"2206.02777","rows_on_this_dataset":6,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":8,"samples_unverified":2,"pointer_only_for_licence":13,"official":{"repos":["idea-research/maskdino"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mask-dino-towards-a-unified-transformer-based-1#ran","syntology_url":"https://syntology.ai/paper/2206.02777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.02777"}}}},{"paper":"/paper/contrastive-learning-rivals-masked-image","slug":"contrastive-learning-rivals-masked-image","title":"Contrastive Learning Rivals Masked Image Modeling in Fine-tuning via Feature Distillation","date":"2022-05-27","arxiv_id":"2205.14141","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":8,"official":{"repos":["SwinTransformer/Feature-Distillation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/contrastive-learning-rivals-masked-image#ran","syntology_url":"https://syntology.ai/paper/2205.14141","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14141"}}}},{"paper":"/paper/revealing-the-dark-secrets-of-masked-image","slug":"revealing-the-dark-secrets-of-masked-image","title":"Revealing the Dark Secrets of Masked Image Modeling","date":"2022-05-26","arxiv_id":"2205.13543","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["SwinTransformer/MIM-Depth-Estimation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/revealing-the-dark-secrets-of-masked-image#ran","syntology_url":"https://syntology.ai/paper/2205.13543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13543"}}}},{"paper":"/paper/photorealistic-text-to-image-diffusion-models","slug":"photorealistic-text-to-image-diffusion-models","title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding","date":"2022-05-23","arxiv_id":"2205.11487","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":5,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":6,"pointer_only_for_licence":9,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/photorealistic-text-to-image-diffusion-models#ran","syntology_url":"https://syntology.ai/paper/2205.11487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11487"}}}},{"paper":"/paper/aggpose-deep-aggregation-vision-transformer","slug":"aggpose-deep-aggregation-vision-transformer","title":"AggPose: Deep Aggregation Vision Transformer for Infant Pose Estimation","date":"2022-05-11","arxiv_id":"2205.05277","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":5,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":7,"official":{"repos":["szar-lab/aggpose"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/aggpose-deep-aggregation-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.05277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05277"}}}},{"paper":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":10,"samples_constructed":5,"samples_ran_checked":10,"samples_ran_instrument_failed":0,"samples_unverified":7,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}}}},{"paper":"/paper/lite-pose-efficient-architecture-design-for","slug":"lite-pose-efficient-architecture-design-for","title":"Lite Pose: Efficient Architecture Design for 2D Human Pose Estimation","date":"2022-05-03","arxiv_id":"2205.01271","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["mit-han-lab/litepose"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lite-pose-efficient-architecture-design-for#ran","syntology_url":"https://syntology.ai/paper/2205.01271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01271"}}}},{"paper":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":24,"samples_ran":18,"samples_constructed":6,"samples_ran_checked":12,"samples_ran_instrument_failed":6,"samples_unverified":6,"pointer_only_for_licence":8,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}}}},{"paper":"/paper/cogview2-faster-and-better-text-to-image","slug":"cogview2-faster-and-better-text-to-image","title":"CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers","date":"2022-04-28","arxiv_id":"2204.14217","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["thudm/cogview2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogview2-faster-and-better-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2204.14217","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14217"}}}},{"paper":"/paper/vitpose-simple-vision-transformer-baselines","slug":"vitpose-simple-vision-transformer-baselines","title":"ViTPose: Simple Vision Transformer Baselines for Human Pose Estimation","date":"2022-04-26","arxiv_id":"2204.12484","rows_on_this_dataset":2,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":31,"samples_ran":18,"samples_constructed":8,"samples_ran_checked":12,"samples_ran_instrument_failed":6,"samples_unverified":13,"pointer_only_for_licence":6,"official":{"repos":["vitae-transformer/vitpose"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vitpose-simple-vision-transformer-baselines#ran","syntology_url":"https://syntology.ai/paper/2204.12484","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.12484"}}}},{"paper":"/paper/hierarchical-text-conditional-image","slug":"hierarchical-text-conditional-image","title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","date":"2022-04-13","arxiv_id":"2204.06125","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":38,"samples_ran":33,"samples_constructed":7,"samples_ran_checked":26,"samples_ran_instrument_failed":7,"samples_unverified":5,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hierarchical-text-conditional-image#ran","syntology_url":"https://syntology.ai/paper/2204.06125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.06125"}}}},{"paper":"/paper/davit-dual-attention-vision-transformers","slug":"davit-dual-attention-vision-transformers","title":"DaViT: Dual Attention Vision Transformers","date":"2022-04-07","arxiv_id":"2204.03645","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":8,"samples_constructed":5,"samples_ran_checked":6,"samples_ran_instrument_failed":2,"samples_unverified":7,"pointer_only_for_licence":0,"official":{"repos":["dingmyu/davit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/davit-dual-attention-vision-transformers#ran","syntology_url":"https://syntology.ai/paper/2204.03645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.03645"}}}},{"paper":"/paper/maxvit-multi-axis-vision-transformer","slug":"maxvit-multi-axis-vision-transformer","title":"MaxViT: Multi-Axis Vision Transformer","date":"2022-04-04","arxiv_id":"2204.01697","rows_on_this_dataset":3,"code_links":15,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":53,"samples_ran":37,"samples_constructed":15,"samples_ran_checked":23,"samples_ran_instrument_failed":14,"samples_unverified":16,"pointer_only_for_licence":9,"official":{"repos":["google-research/maxvit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/maxvit-multi-axis-vision-transformer#ran","syntology_url":"https://syntology.ai/paper/2204.01697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.01697"}}}},{"paper":"/paper/pp-yoloe-an-evolved-version-of-yolo","slug":"pp-yoloe-an-evolved-version-of-yolo","title":"PP-YOLOE: An evolved version of YOLO","date":"2022-03-30","arxiv_id":"2203.16250","rows_on_this_dataset":9,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":27,"samples_ran":22,"samples_constructed":0,"samples_ran_checked":21,"samples_ran_instrument_failed":1,"samples_unverified":5,"pointer_only_for_licence":1,"official":{"repos":["PaddlePaddle/PaddleDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/pp-yoloe-an-evolved-version-of-yolo#ran","syntology_url":"https://syntology.ai/paper/2203.16250","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16250"}}}},{"paper":"/paper/exploring-plain-vision-transformer-backbones","slug":"exploring-plain-vision-transformer-backbones","title":"Exploring Plain Vision Transformer Backbones for Object Detection","date":"2022-03-30","arxiv_id":"2203.16527","rows_on_this_dataset":4,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["facebookresearch/detectron2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/exploring-plain-vision-transformer-backbones#ran","syntology_url":"https://syntology.ai/paper/2203.16527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16527"}}}},{"paper":"/paper/make-a-scene-scene-based-text-to-image","slug":"make-a-scene-scene-based-text-to-image","title":"Make-A-Scene: Scene-Based Text-to-Image Generation with Human Priors","date":"2022-03-24","arxiv_id":"2203.13131","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":2,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/make-a-scene-scene-based-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2203.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13131"}}}},{"paper":"/paper/focal-modulation-networks","slug":"focal-modulation-networks","title":"Focal Modulation Networks","date":"2022-03-22","arxiv_id":"2203.11926","rows_on_this_dataset":5,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/FocalNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/focal-modulation-networks#ran","syntology_url":"https://syntology.ai/paper/2203.11926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.11926"}}}},{"paper":"/paper/e2ec-an-end-to-end-contour-based-method-for","slug":"e2ec-an-end-to-end-contour-based-method-for","title":"E2EC: An End-to-End Contour-based Method for High-Quality High-Speed Instance Segmentation","date":"2022-03-08","arxiv_id":"2203.04074","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":10,"official":{"repos":["zhang-tao-whu/e2ec"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/e2ec-an-end-to-end-contour-based-method-for#ran","syntology_url":"https://syntology.ai/paper/2203.04074","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.04074"}}}},{"paper":"/paper/dino-detr-with-improved-denoising-anchor-1","slug":"dino-detr-with-improved-denoising-anchor-1","title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection","date":"2022-03-07","arxiv_id":"2203.03605","rows_on_this_dataset":4,"code_links":16,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":4,"samples_unverified":5,"pointer_only_for_licence":5,"official":{"repos":["IDEACVR/DINO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dino-detr-with-improved-denoising-anchor-1#ran","syntology_url":"https://syntology.ai/paper/2203.03605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03605"}}}},{"paper":"/paper/dn-detr-accelerate-detr-training-by","slug":"dn-detr-accelerate-detr-training-by","title":"DN-DETR: Accelerate DETR Training by Introducing Query DeNoising","date":"2022-03-02","arxiv_id":"2203.01305","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":22,"samples_ran":17,"samples_constructed":3,"samples_ran_checked":6,"samples_ran_instrument_failed":11,"samples_unverified":5,"pointer_only_for_licence":9,"official":{"repos":["IDEA-Research/detrex","fengli-ust/dn-detr","idea-research/dn-detr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dn-detr-accelerate-detr-training-by#ran","syntology_url":"https://syntology.ai/paper/2203.01305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.01305"}}}},{"paper":"/paper/self-supervised-transformers-for-unsupervised","slug":"self-supervised-transformers-for-unsupervised","title":"Self-Supervised Transformers for Unsupervised Object Discovery using Normalized Cut","date":"2022-02-23","arxiv_id":"2202.11539","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-transformers-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2202.11539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.11539"}}}},{"paper":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}}}},{"paper":"/paper/visual-attention-network","slug":"visual-attention-network","title":"Visual Attention Network","date":"2022-02-20","arxiv_id":"2202.09741","rows_on_this_dataset":2,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["Visual-Attention-Network/VAN-Classification"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/visual-attention-network#ran","syntology_url":"https://syntology.ai/paper/2202.09741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.09741"}}}}],"record_sha256":"4169e88d93f05fc60995125faae947ea93625bbf3cbba7cf8b9c3fde683bbfb3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}