{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/coco/papers/6","list_of":"/dataset/coco","dataset":"COCO (Common Objects in Context)","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"archive","order_definition":"date (newest first), then slug","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":6,"pages_in_order":6,"rows_per_page":100,"rows":[501,579],"of":579,"counts":{"papers_with_a_benchmark_row":579,"with_a_code_link":504,"where_syntology_ran_a_sample":256,"not_listed_spam_title":0,"listed":579,"listed_where_code_ran":256,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":226,"every_run_a_failure_of_syntologys_instrument":30,"listed_with_a_run_with_no_instrument_failure":226,"listed_every_run_a_failure_of_syntologys_instrument":30,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/coco","prev":"/dataset/coco/papers/5","next":null,"papers":[{"paper":"/paper/polarity-loss-for-zero-shot-object-detection","slug":"polarity-loss-for-zero-shot-object-detection","title":"Polarity Loss for Zero-shot Object Detection","date":"2018-11-22","arxiv_id":"1811.08982","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/rethinking-imagenet-pre-training","slug":"rethinking-imagenet-pre-training","title":"Rethinking ImageNet Pre-training","date":"2018-11-21","arxiv_id":"1811.08883","rows_on_this_dataset":3,"code_links":1,"syntology":null},{"paper":"/paper/gradient-harmonized-single-stage-detector","slug":"gradient-harmonized-single-stage-detector","title":"Gradient Harmonized Single-stage Detector","date":"2018-11-13","arxiv_id":"1811.05181","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["libuyu/GHM_Detection"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/gradient-harmonized-single-stage-detector#ran","syntology_url":"https://syntology.ai/paper/1811.05181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.05181"}}}},{"paper":"/paper/m2det-a-single-shot-object-detector-based-on","slug":"m2det-a-single-shot-object-detector-based-on","title":"M2Det: A Single-Shot Object Detector based on Multi-Level Feature Pyramid Network","date":"2018-11-12","arxiv_id":"1811.04533","rows_on_this_dataset":6,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/m2det-a-single-shot-object-detector-based-on#ran","syntology_url":"https://syntology.ai/paper/1811.04533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.04533"}}}},{"paper":"/paper/softer-nms-rethinking-bounding-box-regression","slug":"softer-nms-rethinking-bounding-box-regression","title":"Bounding Box Regression with Uncertainty for Accurate Object Detection","date":"2018-09-23","arxiv_id":"1809.08545","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["yihui-he/softer-NMS","yihui-he/KL-Loss"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/softer-nms-rethinking-bounding-box-regression#ran","syntology_url":"https://syntology.ai/paper/1809.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.08545"}}}},{"paper":"/paper/panoptic-segmentation-with-a-joint-semantic","slug":"panoptic-segmentation-with-a-joint-semantic","title":"Panoptic Segmentation with a Joint Semantic and Instance Segmentation Network","date":"2018-09-06","arxiv_id":"1809.02110","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/associating-inter-image-salient-instances-for","slug":"associating-inter-image-salient-instances-for","title":"Associating Inter-Image Salient Instances for Weakly Supervised Semantic Segmentation","date":"2018-09-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/multimodal-differential-network-for-visual","slug":"multimodal-differential-network-for-visual","title":"Multimodal Differential Network for Visual Question Generation","date":"2018-08-12","arxiv_id":"1808.03986","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/question-guided-hybrid-convolution-for-visual","slug":"question-guided-hybrid-convolution-for-visual","title":"Question-Guided Hybrid Convolution for Visual Question Answering","date":"2018-08-08","arxiv_id":"1808.02632","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cornernet-detecting-objects-as-paired","slug":"cornernet-detecting-objects-as-paired","title":"CornerNet: Detecting Objects as Paired Keypoints","date":"2018-08-03","arxiv_id":"1808.01244","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":8,"pointer_only_for_licence":2,"official":{"repos":["princeton-vl/CornerNet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cornernet-detecting-objects-as-paired#ran","syntology_url":"https://syntology.ai/paper/1808.01244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.01244"}}}},{"paper":"/paper/acquisition-of-localization-confidence-for","slug":"acquisition-of-localization-confidence-for","title":"Acquisition of Localization Confidence for Accurate Object Detection","date":"2018-07-30","arxiv_id":"1807.11590","rows_on_this_dataset":1,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":17,"samples_constructed":0,"samples_ran_checked":15,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["vacancy/PreciseRoIPooling"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/acquisition-of-localization-confidence-for#ran","syntology_url":"https://syntology.ai/paper/1807.11590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.11590"}}}},{"paper":"/paper/where-are-the-blobs-counting-by-localization","slug":"where-are-the-blobs-counting-by-localization","title":"Where are the Blobs: Counting by Localization with Point Supervision","date":"2018-07-25","arxiv_id":"1807.09856","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":1,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/where-are-the-blobs-counting-by-localization#ran","syntology_url":"https://syntology.ai/paper/1807.09856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.09856"}}}},{"paper":"/paper/invariant-information-distillation-for","slug":"invariant-information-distillation-for","title":"Invariant Information Clustering for Unsupervised Image Classification and Segmentation","date":"2018-07-17","arxiv_id":"1807.06653","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":2,"samples_unverified":7,"pointer_only_for_licence":0,"official":{"repos":["xu-ji/IIC"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/invariant-information-distillation-for#ran","syntology_url":"https://syntology.ai/paper/1807.06653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06653"}}}},{"paper":"/paper/multiposenet-fast-multi-person-pose","slug":"multiposenet-fast-multi-person-pose","title":"MultiPoseNet: Fast Multi-Person Pose Estimation using Pose Residual Network","date":"2018-07-11","arxiv_id":"1807.04067","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/r-vqa-learning-visual-relation-facts-with","slug":"r-vqa-learning-visual-relation-facts-with","title":"R-VQA: Learning Visual Relation Facts with Semantic Attention for Visual Question Answering","date":"2018-05-24","arxiv_id":"1805.09701","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/sniper-efficient-multi-scale-training","slug":"sniper-efficient-multi-scale-training","title":"SNIPER: Efficient Multi-Scale Training","date":"2018-05-23","arxiv_id":"1805.09300","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/simple-baselines-for-human-pose-estimation","slug":"simple-baselines-for-human-pose-estimation","title":"Simple Baselines for Human Pose Estimation and Tracking","date":"2018-04-17","arxiv_id":"1804.06208","rows_on_this_dataset":5,"code_links":27,"syntology":null},{"paper":"/paper/yolov3-an-incremental-improvement","slug":"yolov3-an-incremental-improvement","title":"YOLOv3: An Incremental Improvement","date":"2018-04-08","arxiv_id":"1804.02767","rows_on_this_dataset":1,"code_links":311,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":124,"samples_ran":94,"samples_constructed":0,"samples_ran_checked":83,"samples_ran_instrument_failed":11,"samples_unverified":30,"pointer_only_for_licence":24,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/yolov3-an-incremental-improvement#ran","syntology_url":"https://syntology.ai/paper/1804.02767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1804.02767"}}}},{"paper":"/paper/group-normalization","slug":"group-normalization","title":"Group Normalization","date":"2018-03-22","arxiv_id":"1803.08494","rows_on_this_dataset":3,"code_links":22,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":7,"samples_constructed":2,"samples_ran_checked":5,"samples_ran_instrument_failed":2,"samples_unverified":8,"pointer_only_for_licence":5,"official":{"repos":["ppwwyyxx/GroupNorm-reproduce"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/group-normalization#ran","syntology_url":"https://syntology.ai/paper/1803.08494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.08494"}}}},{"paper":"/paper/personlab-person-pose-estimation-and-instance","slug":"personlab-person-pose-estimation-and-instance","title":"PersonLab: Person Pose Estimation and Instance Segmentation with a Bottom-Up, Part-Based, Geometric Embedding Model","date":"2018-03-22","arxiv_id":"1803.08225","rows_on_this_dataset":2,"code_links":3,"syntology":null},{"paper":"/paper/stacked-cross-attention-for-image-text","slug":"stacked-cross-attention-for-image-text","title":"Stacked Cross Attention for Image-Text Matching","date":"2018-03-21","arxiv_id":"1803.08024","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":16,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":6,"samples_unverified":3,"pointer_only_for_licence":1,"official":{"repos":["kuanghuei/SCAN"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/stacked-cross-attention-for-image-text#ran","syntology_url":"https://syntology.ai/paper/1803.08024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.08024"}}}},{"paper":"/paper/lstd-a-low-shot-transfer-detector-for-object","slug":"lstd-a-low-shot-transfer-detector-for-object","title":"LSTD: A Low-Shot Transfer Detector for Object Detection","date":"2018-03-05","arxiv_id":"1803.01529","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/path-aggregation-network-for-instance","slug":"path-aggregation-network-for-instance","title":"Path Aggregation Network for Instance Segmentation","date":"2018-03-05","arxiv_id":"1803.01534","rows_on_this_dataset":3,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["ShuLiu1993/PANet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/path-aggregation-network-for-instance#ran","syntology_url":"https://syntology.ai/paper/1803.01534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.01534"}}}},{"paper":"/paper/chatpainter-improving-text-to-image","slug":"chatpainter-improving-text-to-image","title":"ChatPainter: Improving Text to Image Generation using Dialogue","date":"2018-02-22","arxiv_id":"1802.08216","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/pose-flow-efficient-online-pose-tracking","slug":"pose-flow-efficient-online-pose-tracking","title":"Pose Flow: Efficient Online Pose Tracking","date":"2018-02-03","arxiv_id":"1802.00977","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/detect-and-track-efficient-pose-estimation-in","slug":"detect-and-track-efficient-pose-estimation-in","title":"Detect-and-Track: Efficient Pose Estimation in Videos","date":"2017-12-26","arxiv_id":"1712.09184","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/masklab-instance-segmentation-by-refining","slug":"masklab-instance-segmentation-by-refining","title":"MaskLab: Instance Segmentation by Refining Object Detection with Semantic and Direction Features","date":"2017-12-13","arxiv_id":"1712.04837","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/learning-semantic-concepts-and-order-for","slug":"learning-semantic-concepts-and-order-for","title":"Learning Semantic Concepts and Order for Image and Sentence Matching","date":"2017-12-06","arxiv_id":"1712.02036","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cascade-r-cnn-delving-into-high-quality","slug":"cascade-r-cnn-delving-into-high-quality","title":"Cascade R-CNN: Delving into High Quality Object Detection","date":"2017-12-03","arxiv_id":"1712.00726","rows_on_this_dataset":6,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["zhaoweicai/cascade-rcnn"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cascade-r-cnn-delving-into-high-quality#ran","syntology_url":"https://syntology.ai/paper/1712.00726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.00726"}}}},{"paper":"/paper/attngan-fine-grained-text-to-image-generation","slug":"attngan-fine-grained-text-to-image-generation","title":"AttnGAN: Fine-Grained Text to Image Generation with Attentional Generative Adversarial Networks","date":"2017-11-28","arxiv_id":"1711.10485","rows_on_this_dataset":1,"code_links":20,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/attngan-fine-grained-text-to-image-generation#ran","syntology_url":"https://syntology.ai/paper/1711.10485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.10485"}}}},{"paper":"/paper/an-analysis-of-scale-invariance-in-object-1","slug":"an-analysis-of-scale-invariance-in-object-1","title":"An Analysis of Scale Invariance in Object Detection - SNIP","date":"2017-11-22","arxiv_id":"1711.08189","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-object-discovery-by","slug":"weakly-supervised-object-discovery-by","title":"Weakly Supervised Object Discovery by Generative Adversarial & Ranking Networks","date":"2017-11-22","arxiv_id":"1711.08174","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/non-local-neural-networks","slug":"non-local-neural-networks","title":"Non-local Neural Networks","date":"2017-11-21","arxiv_id":"1711.07971","rows_on_this_dataset":7,"code_links":32,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":4,"official":{"repos":["facebookresearch/video-nonlocal-net"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/non-local-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1711.07971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.07971"}}}},{"paper":"/paper/cascaded-pyramid-network-for-multi-person","slug":"cascaded-pyramid-network-for-multi-person","title":"Cascaded Pyramid Network for Multi-Person Pose Estimation","date":"2017-11-20","arxiv_id":"1711.07319","rows_on_this_dataset":7,"code_links":5,"syntology":null},{"paper":"/paper/co-attending-free-form-regions-and-detections","slug":"co-attending-free-form-regions-and-detections","title":"Co-attending Free-form Regions and Detections with Multi-modal Multiplicative Feature Embedding for Visual Question Answering","date":"2017-11-18","arxiv_id":"1711.06794","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/single-shot-refinement-neural-network-for","slug":"single-shot-refinement-neural-network-for","title":"Single-Shot Refinement Neural Network for Object Detection","date":"2017-11-18","arxiv_id":"1711.06897","rows_on_this_dataset":3,"code_links":16,"syntology":null},{"paper":"/paper/dual-path-convolutional-image-text-embedding","slug":"dual-path-convolutional-image-text-embedding","title":"Dual-Path Convolutional Image-Text Embeddings with Instance Loss","date":"2017-11-15","arxiv_id":"1711.05535","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/high-order-attention-models-for-visual","slug":"high-order-attention-models-for-visual","title":"High-Order Attention Models for Visual Question Answering","date":"2017-11-12","arxiv_id":"1711.04323","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/stackgan-realistic-image-synthesis-with","slug":"stackgan-realistic-image-synthesis-with","title":"StackGAN++: Realistic Image Synthesis with Stacked Generative Adversarial Networks","date":"2017-10-19","arxiv_id":"1710.10916","rows_on_this_dataset":1,"code_links":16,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":31,"samples_ran":22,"samples_constructed":0,"samples_ran_checked":14,"samples_ran_instrument_failed":8,"samples_unverified":9,"pointer_only_for_licence":3,"official":{"repos":["hanzhanggit/StackGAN","hanzhanggit/StackGAN-v2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/stackgan-realistic-image-synthesis-with#ran","syntology_url":"https://syntology.ai/paper/1710.10916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.10916"}}}},{"paper":"/paper/soft-proposal-networks-for-weakly-supervised","slug":"soft-proposal-networks-for-weakly-supervised","title":"Soft Proposal Networks for Weakly Supervised Object Localization","date":"2017-09-06","arxiv_id":"1709.01829","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/chainercv-a-library-for-deep-learning-in","slug":"chainercv-a-library-for-deep-learning-in","title":"ChainerCV: a Library for Deep Learning in Computer Vision","date":"2017-08-28","arxiv_id":"1708.08169","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/focal-loss-for-dense-object-detection","slug":"focal-loss-for-dense-object-detection","title":"Focal Loss for Dense Object Detection","date":"2017-08-07","arxiv_id":"1708.02002","rows_on_this_dataset":2,"code_links":234,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":9,"samples_unverified":0,"pointer_only_for_licence":6,"official":{"repos":["facebookresearch/detectron"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/focal-loss-for-dense-object-detection#ran","syntology_url":"https://syntology.ai/paper/1708.02002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1708.02002"}}}},{"paper":"/paper/unsupervised-object-discovery-and-co","slug":"unsupervised-object-discovery-and-co","title":"Unsupervised Object Discovery and Co-Localization by Deep Descriptor Transforming","date":"2017-07-20","arxiv_id":"1707.06397","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/revisiting-unreasonable-effectiveness-of-data","slug":"revisiting-unreasonable-effectiveness-of-data","title":"Revisiting Unreasonable Effectiveness of Data in Deep Learning Era","date":"2017-07-10","arxiv_id":"1707.02968","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/wildcat-weakly-supervised-learning-of-deep","slug":"wildcat-weakly-supervised-learning-of-deep","title":"WILDCAT: Weakly Supervised Learning of Deep ConvNets for Image Classification, Pointwise Localization and Segmentation","date":"2017-07-01","arxiv_id":null,"rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/few-example-object-detection-with-model","slug":"few-example-object-detection-with-model","title":"Few-Example Object Detection with Model Communication","date":"2017-06-26","arxiv_id":"1706.08249","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/mask-r-cnn","slug":"mask-r-cnn","title":"Mask R-CNN","date":"2017-03-20","arxiv_id":"1703.06870","rows_on_this_dataset":9,"code_links":179,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":140,"samples_ran":101,"samples_constructed":3,"samples_ran_checked":90,"samples_ran_instrument_failed":11,"samples_unverified":39,"pointer_only_for_licence":32,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mask-r-cnn#ran","syntology_url":"https://syntology.ai/paper/1703.06870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.06870"}}}},{"paper":"/paper/deformable-convolutional-networks","slug":"deformable-convolutional-networks","title":"Deformable Convolutional Networks","date":"2017-03-17","arxiv_id":"1703.06211","rows_on_this_dataset":1,"code_links":38,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":4,"official":{"repos":["msracver/Deformable-ConvNets"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deformable-convolutional-networks#ran","syntology_url":"https://syntology.ai/paper/1703.06211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.06211"}}}},{"paper":"/paper/learning-spatial-regularization-with-image","slug":"learning-spatial-regularization-with-image","title":"Learning Spatial Regularization with Image-level Supervisions for Multi-label Image Classification","date":"2017-02-20","arxiv_id":"1702.05891","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/towards-accurate-multi-person-pose-estimation","slug":"towards-accurate-multi-person-pose-estimation","title":"Towards Accurate Multi-person Pose Estimation in the Wild","date":"2017-01-06","arxiv_id":"1701.01779","rows_on_this_dataset":6,"code_links":0,"syntology":null},{"paper":"/paper/beyond-skip-connections-top-down-modulation","slug":"beyond-skip-connections-top-down-modulation","title":"Beyond Skip Connections: Top-Down Modulation for Object Detection","date":"2016-12-20","arxiv_id":"1612.06851","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/feature-pyramid-networks-for-object-detection","slug":"feature-pyramid-networks-for-object-detection","title":"Feature Pyramid Networks for Object Detection","date":"2016-12-09","arxiv_id":"1612.03144","rows_on_this_dataset":2,"code_links":85,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":51,"samples_ran":33,"samples_constructed":9,"samples_ran_checked":31,"samples_ran_instrument_failed":2,"samples_unverified":18,"pointer_only_for_licence":11,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/feature-pyramid-networks-for-object-detection#ran","syntology_url":"https://syntology.ai/paper/1612.03144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.03144"}}}},{"paper":"/paper/making-the-v-in-vqa-matter-elevating-the-role","slug":"making-the-v-in-vqa-matter-elevating-the-role","title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering","date":"2016-12-02","arxiv_id":"1612.00837","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/rmpe-regional-multi-person-pose-estimation","slug":"rmpe-regional-multi-person-pose-estimation","title":"RMPE: Regional Multi-person Pose Estimation","date":"2016-12-01","arxiv_id":"1612.00137","rows_on_this_dataset":5,"code_links":14,"syntology":null},{"paper":"/paper/realtime-multi-person-2d-pose-estimation","slug":"realtime-multi-person-2d-pose-estimation","title":"Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","date":"2016-11-24","arxiv_id":"1611.08050","rows_on_this_dataset":3,"code_links":61,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":23,"samples_ran":18,"samples_constructed":0,"samples_ran_checked":16,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":4,"official":{"repos":["ZheC/Realtime_Multi-Person_Pose_Estimation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/realtime-multi-person-2d-pose-estimation#ran","syntology_url":"https://syntology.ai/paper/1611.08050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.08050"}}}},{"paper":"/paper/weakly-supervised-cascaded-convolutional","slug":"weakly-supervised-cascaded-convolutional","title":"Weakly Supervised Cascaded Convolutional Networks","date":"2016-11-24","arxiv_id":"1611.08258","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/fully-convolutional-instance-aware-semantic","slug":"fully-convolutional-instance-aware-semantic","title":"Fully Convolutional Instance-aware Semantic Segmentation","date":"2016-11-23","arxiv_id":"1611.07709","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":4,"official":{"repos":["daijifeng001/TA-FCN"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/fully-convolutional-instance-aware-semantic#ran","syntology_url":"https://syntology.ai/paper/1611.07709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.07709"}}}},{"paper":"/paper/associative-embedding-end-to-end-learning-for","slug":"associative-embedding-end-to-end-learning-for","title":"Associative Embedding: End-to-End Learning for Joint Detection and Grouping","date":"2016-11-16","arxiv_id":"1611.05424","rows_on_this_dataset":3,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":3,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/associative-embedding-end-to-end-learning-for#ran","syntology_url":"https://syntology.ai/paper/1611.05424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.05424"}}}},{"paper":"/paper/graph-structured-representations-for-visual","slug":"graph-structured-representations-for-visual","title":"Graph-Structured Representations for Visual Question Answering","date":"2016-09-19","arxiv_id":"1609.05600","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/training-recurrent-answering-units-with-joint","slug":"training-recurrent-answering-units-with-joint","title":"Training Recurrent Answering Units with Joint Loss Minimization for VQA","date":"2016-06-12","arxiv_id":"1606.03647","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/multimodal-compact-bilinear-pooling-for","slug":"multimodal-compact-bilinear-pooling-for","title":"Multimodal Compact Bilinear Pooling for Visual Question Answering and Visual Grounding","date":"2016-06-06","arxiv_id":"1606.01847","rows_on_this_dataset":2,"code_links":10,"syntology":null},{"paper":"/paper/multimodal-residual-learning-for-visual-qa","slug":"multimodal-residual-learning-for-visual-qa","title":"Multimodal Residual Learning for Visual QA","date":"2016-06-05","arxiv_id":"1606.01455","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/hierarchical-question-image-co-attention-for","slug":"hierarchical-question-image-co-attention-for","title":"Hierarchical Question-Image Co-Attention for Visual Question Answering","date":"2016-05-31","arxiv_id":"1606.00061","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":1,"official":{"repos":["jiasenlu/HieCoAttenVQA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hierarchical-question-image-co-attention-for#ran","syntology_url":"https://syntology.ai/paper/1606.00061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.00061"}}}},{"paper":"/paper/counting-everyday-objects-in-everyday-scenes","slug":"counting-everyday-objects-in-everyday-scenes","title":"Counting Everyday Objects in Everyday Scenes","date":"2016-04-12","arxiv_id":"1604.03505","rows_on_this_dataset":5,"code_links":1,"syntology":null},{"paper":"/paper/a-multipath-network-for-object-detection","slug":"a-multipath-network-for-object-detection","title":"A MultiPath Network for Object Detection","date":"2016-04-07","arxiv_id":"1604.02135","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/a-focused-dynamic-attention-model-for-visual","slug":"a-focused-dynamic-attention-model-for-visual","title":"A Focused Dynamic Attention Model for Visual Question Answering","date":"2016-04-06","arxiv_id":"1604.01485","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/image-captioning-and-visual-question","slug":"image-captioning-and-visual-question","title":"Image Captioning and Visual Question Answering Based on Attributes and External Knowledge","date":"2016-03-09","arxiv_id":"1603.02814","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/dynamic-memory-networks-for-visual-and","slug":"dynamic-memory-networks-for-visual-and","title":"Dynamic Memory Networks for Visual and Textual Question Answering","date":"2016-03-04","arxiv_id":"1603.01417","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":7,"samples_unverified":0,"pointer_only_for_licence":7,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dynamic-memory-networks-for-visual-and#ran","syntology_url":"https://syntology.ai/paper/1603.01417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1603.01417"}}}},{"paper":"/paper/weakly-supervised-localization-using-deep","slug":"weakly-supervised-localization-using-deep","title":"Weakly Supervised Localization using Deep Feature Maps","date":"2016-03-01","arxiv_id":"1603.00489","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/instance-aware-semantic-segmentation-via","slug":"instance-aware-semantic-segmentation-via","title":"Instance-aware Semantic Segmentation via Multi-task Network Cascades","date":"2015-12-14","arxiv_id":"1512.04412","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/deep-residual-learning-for-image-recognition","slug":"deep-residual-learning-for-image-recognition","title":"Deep Residual Learning for Image Recognition","date":"2015-12-10","arxiv_id":"1512.03385","rows_on_this_dataset":3,"code_links":484,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":377,"samples_ran":254,"samples_constructed":108,"samples_ran_checked":166,"samples_ran_instrument_failed":88,"samples_unverified":123,"pointer_only_for_licence":193,"official":{"repos":["KaimingHe/resnet-1k-layers"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deep-residual-learning-for-image-recognition#ran","syntology_url":"https://syntology.ai/paper/1512.03385","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1512.03385"}}}},{"paper":"/paper/neural-self-talk-image-understanding-via","slug":"neural-self-talk-image-understanding-via","title":"Neural Self Talk: Image Understanding via Continuous Questioning and Answering","date":"2015-12-10","arxiv_id":"1512.03460","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/simple-baseline-for-visual-question-answering","slug":"simple-baseline-for-visual-question-answering","title":"Simple Baseline for Visual Question Answering","date":"2015-12-07","arxiv_id":"1512.02167","rows_on_this_dataset":2,"code_links":7,"syntology":null},{"paper":"/paper/ask-attend-and-answer-exploring-question","slug":"ask-attend-and-answer-exploring-question","title":"Ask, Attend and Answer: Exploring Question-Guided Spatial Attention for Visual Question Answering","date":"2015-11-17","arxiv_id":"1511.05234","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/pronet-learning-to-propose-object-specific","slug":"pronet-learning-to-propose-object-specific","title":"ProNet: Learning to Propose Object-specific Boxes for Cascaded Neural Networks","date":"2015-11-12","arxiv_id":"1511.03776","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-deep-detection-networks","slug":"weakly-supervised-deep-detection-networks","title":"Weakly Supervised Deep Detection Networks","date":"2015-11-09","arxiv_id":"1511.02853","rows_on_this_dataset":1,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["hbilen/WSDDN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/weakly-supervised-deep-detection-networks#ran","syntology_url":"https://syntology.ai/paper/1511.02853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.02853"}}}},{"paper":"/paper/stacked-attention-networks-for-image-question","slug":"stacked-attention-networks-for-image-question","title":"Stacked Attention Networks for Image Question Answering","date":"2015-11-07","arxiv_id":"1511.02274","rows_on_this_dataset":1,"code_links":16,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/stacked-attention-networks-for-image-question#ran","syntology_url":"https://syntology.ai/paper/1511.02274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1511.02274"}}}},{"paper":"/paper/vqa-visual-question-answering","slug":"vqa-visual-question-answering","title":"VQA: Visual Question Answering","date":"2015-05-03","arxiv_id":"1505.00468","rows_on_this_dataset":10,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":6,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vqa-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/1505.00468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1505.00468"}}}},{"paper":"/paper/deep-visual-semantic-alignments-for","slug":"deep-visual-semantic-alignments-for","title":"Deep Visual-Semantic Alignments for Generating Image Descriptions","date":"2014-12-07","arxiv_id":"1412.2306","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deep-visual-semantic-alignments-for#ran","syntology_url":"https://syntology.ai/paper/1412.2306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1412.2306"}}}}],"record_sha256":"27dd1ed1ff66f7b570985fd4982b89ca91006338ad031f6869f8976b2850d21f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}