{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/spatial-transformer/papers/2","list_of":"/method/spatial-transformer","method":"Spatial Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,169],"of":169,"counts":{"archive_papers_tagged":169,"with_a_code_link":73,"where_syntology_ran_a_sample":12,"not_listed_spam_title":0,"listed":169,"listed_where_code_ran":12,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":10,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":10,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/spatial-transformer","prev":"/method/spatial-transformer","next":null,"papers":[{"paper":"/paper/unmasking-the-inductive-biases-of","slug":"unmasking-the-inductive-biases-of","title":"Benchmarking Unsupervised Object Representations for Video Sequences","date":"2020-06-12","arxiv_id":"2006.07034","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ecker-lab/object-centric-representation-benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/end-to-end-learning-for-semiquantitative","slug":"end-to-end-learning-for-semiquantitative","title":"BS-Net: learning COVID-19 pneumonia severity on a large Chest X-Ray dataset","date":"2020-06-08","arxiv_id":"2006.04603","n_code_links":2,"syntology":null},{"paper":null,"slug":"rdcface-radial-distortion-correction-for-face","title":"RDCFace: Radial Distortion Correction for Face Recognition","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-sparse-view-backprojection-via","title":"Unsupervised Sparse-view Backprojection via Convolutional and Spatial Transformer Networks","date":"2020-06-01","arxiv_id":"2006.01658","n_code_links":0,"syntology":null},{"paper":"/paper/a-sim2real-deep-learning-approach-for-the","slug":"a-sim2real-deep-learning-approach-for-the","title":"A Sim2Real Deep Learning Approach for the Transformation of Images from Multiple Vehicle-Mounted Cameras to a Semantically Segmented Image in Bird's Eye View","date":"2020-05-08","arxiv_id":"2005.04078","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ika-rwth-aachen/Cam2BEV"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/image-morphing-with-perceptual-constraints","slug":"image-morphing-with-perceptual-constraints","title":"Image Morphing with Perceptual Constraints and STN Alignment","date":"2020-04-29","arxiv_id":"2004.14071","n_code_links":1,"syntology":null},{"paper":"/paper/cortical-surface-registration-using","slug":"cortical-surface-registration-using","title":"Cortical surface registration using unsupervised learning","date":"2020-04-09","arxiv_id":"2004.04617","n_code_links":1,"syntology":null},{"paper":"/paper/probabilistic-spatial-transformers-for","slug":"probabilistic-spatial-transformers-for","title":"Probabilistic Spatial Transformer Networks","date":"2020-04-07","arxiv_id":"2004.03637","n_code_links":1,"syntology":null},{"paper":"/paper/autotoon-automatic-geometric-warping-for-face","slug":"autotoon-automatic-geometric-warping-for-face","title":"AutoToon: Automatic Geometric Warping for Face Cartoon Generation","date":"2020-04-06","arxiv_id":"2004.02377","n_code_links":1,"syntology":null},{"paper":"/paper/lidar-based-online-3d-video-object-detection","slug":"lidar-based-online-3d-video-object-detection","title":"LiDAR-based Online 3D Video Object Detection with Graph-based Message Passing and Spatiotemporal Transformer Attention","date":"2020-04-03","arxiv_id":"2004.01389","n_code_links":1,"syntology":null},{"paper":"/paper/generalizing-spatial-transformers-to","slug":"generalizing-spatial-transformers-to","title":"Generalizing Spatial Transformers to Projective Geometry with Applications to 2D/3D Registration","date":"2020-03-24","arxiv_id":"2003.10987","n_code_links":1,"syntology":null},{"paper":null,"slug":"detecting-lane-and-road-markings-at-a","title":"Detecting Lane and Road Markings at A Distance with Perspective Transformer Layers","date":"2020-03-19","arxiv_id":"2003.08550","n_code_links":0,"syntology":null},{"paper":null,"slug":"disease-detection-from-lung-x-ray-images","title":"Hybrid Deep Learning for Detecting Lung Diseases from X-ray Images","date":"2020-03-02","arxiv_id":"2003.00682","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-face-parsing-via-interlinked","slug":"end-to-end-face-parsing-via-interlinked","title":"End-to-End Face Parsing via Interlinked Convolutional Neural Networks","date":"2020-02-12","arxiv_id":"2002.04831","n_code_links":1,"syntology":null},{"paper":null,"slug":"multistage-model-for-robust-face-alignment","title":"Multistage Model for Robust Face Alignment Using Deep Neural Networks","date":"2020-02-04","arxiv_id":"2002.01075","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-a-layout-transfer-network-for","title":"Learning a Layout Transfer Network for Context Aware Object Detection","date":"2019-12-09","arxiv_id":"1912.03865","n_code_links":0,"syntology":null},{"paper":"/paper/disentangle-align-and-fuse-for-multimodal-and","slug":"disentangle-align-and-fuse-for-multimodal-and","title":"Disentangle, align and fuse for multimodal and semi-supervised image segmentation","date":"2019-11-11","arxiv_id":"1911.04417","n_code_links":2,"syntology":null},{"paper":null,"slug":"190909801","title":"Adversarial Learning of General Transformations for Data Augmentation","date":"2019-09-21","arxiv_id":"1909.09801","n_code_links":0,"syntology":null},{"paper":null,"slug":"jointly-aligning-millions-of-images-with-deep","title":"Jointly Aligning Millions of Images with Deep Penalised Reconstruction Congealing","date":"2019-08-12","arxiv_id":"1908.04130","n_code_links":0,"syntology":null},{"paper":"/paper/locality-constrained-spatial-transformer","slug":"locality-constrained-spatial-transformer","title":"Locality-constrained Spatial Transformer Network for Video Crowd Counting","date":"2019-07-18","arxiv_id":"1907.07911","n_code_links":1,"syntology":null},{"paper":"/paper/one-shot-learning-for-deformable-medical","slug":"one-shot-learning-for-deformable-medical","title":"One Shot Learning for Deformable Medical Image Registration and Periodic Motion Tracking","date":"2019-07-10","arxiv_id":"1907.04641","n_code_links":1,"syntology":null},{"paper":"/paper/spatial-transformer-for-3d-points","slug":"spatial-transformer-for-3d-points","title":"Spatial Transformer for 3D Point Clouds","date":"2019-06-26","arxiv_id":"1906.10887","n_code_links":1,"syntology":null},{"paper":"/paper/explicit-disentanglement-of-appearance-and","slug":"explicit-disentanglement-of-appearance-and","title":"Explicit Disentanglement of Appearance and Perspective in Generative Models","date":"2019-06-11","arxiv_id":"1906.11881","n_code_links":1,"syntology":null},{"paper":"/paper/laf-net-locally-adaptive-fusion-networks-for","slug":"laf-net-locally-adaptive-fusion-networks-for","title":"LAF-Net: Locally Adaptive Fusion Networks for Stereo Confidence Estimation","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-caricature-face-parsing","slug":"weakly-supervised-caricature-face-parsing","title":"Weakly-supervised Caricature Face Parsing through Domain Adaptation","date":"2019-05-13","arxiv_id":"1905.05091","n_code_links":1,"syntology":null},{"paper":null,"slug":"identity-preserving-face-recovery-from-1","title":"Identity-preserving Face Recovery from Stylized Portraits","date":"2019-04-07","arxiv_id":"1904.04241","n_code_links":0,"syntology":null},{"paper":null,"slug":"stnreid-deep-convolutional-networks-with","title":"STNReID : Deep Convolutional Networks with Pairwise Spatial Transformer Networks for Partial Person Re-identification","date":"2019-03-17","arxiv_id":"1903.07072","n_code_links":0,"syntology":null},{"paper":null,"slug":"gq-stn-optimizing-one-shot-grasp-detection","title":"GQ-STN: Optimizing One-Shot Grasp Detection based on Robustness Classifier","date":"2019-03-06","arxiv_id":"1903.02489","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-deep-image-clustering-with-spatial","title":"Improving Deep Image Clustering With Spatial Transformer Layers","date":"2019-02-09","arxiv_id":"1902.05401","n_code_links":0,"syntology":null},{"paper":"/paper/linearized-multi-sampling-for-differentiable","slug":"linearized-multi-sampling-for-differentiable","title":"Linearized Multi-Sampling for Differentiable Image Transformation","date":"2019-01-22","arxiv_id":"1901.07124","n_code_links":1,"syntology":null},{"paper":null,"slug":"composite-shape-modeling-via-latent-space","title":"Composite Shape Modeling via Latent Space Factorization","date":"2019-01-09","arxiv_id":"1901.02968","n_code_links":0,"syntology":null},{"paper":null,"slug":"multitask-painting-categorization-by-deep","title":"Multitask Painting Categorization by Deep Multibranch Neural Network","date":"2018-12-19","arxiv_id":"1812.08052","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-transformer-networks-for-semantic","slug":"recurrent-transformer-networks-for-semantic","title":"Recurrent Transformer Networks for Semantic Correspondence","date":"2018-10-29","arxiv_id":"1810.12155","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-zoom-a-saliency-based-sampling","slug":"learning-to-zoom-a-saliency-based-sampling","title":"Learning to Zoom: a Saliency-Based Sampling Layer for Neural Networks","date":"2018-09-10","arxiv_id":"1809.03355","n_code_links":1,"syntology":null},{"paper":null,"slug":"destnet-densely-fused-spatial-transformer","title":"DeSTNet: Densely Fused Spatial Transformer Networks","date":"2018-07-11","arxiv_id":"1807.04050","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-based-hand-pose-estimation-for","title":"Model-based Hand Pose Estimation for Generalized Hand Shape with Appearance Normalization","date":"2018-07-02","arxiv_id":"1807.00898","n_code_links":0,"syntology":null},{"paper":null,"slug":"localization-a-missing-link-in-the-pipeline","title":"Localization: A Missing Link in the Pipeline of Object Matching and Registration","date":"2018-05-01","arxiv_id":"1805.00223","n_code_links":0,"syntology":null},{"paper":null,"slug":"cram-clued-recurrent-attention-model","title":"CRAM: Clued Recurrent Attention Model","date":"2018-04-28","arxiv_id":"1804.10844","n_code_links":0,"syntology":null},{"paper":null,"slug":"statistical-transformer-networks-learning","title":"Statistical transformer networks: learning shape and appearance models via self supervision","date":"2018-04-07","arxiv_id":"1804.02541","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-and-deep-person-re-identification","title":"Efficient and Deep Person Re-Identification using Multi-Level Similarity","date":"2018-03-30","arxiv_id":"1803.11353","n_code_links":0,"syntology":null},{"paper":"/paper/st-gan-spatial-transformer-generative","slug":"st-gan-spatial-transformer-generative","title":"ST-GAN: Spatial Transformer Generative Adversarial Networks for Image Compositing","date":"2018-03-05","arxiv_id":"1803.01837","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["chenhsuanlin/spatial-transformer-GAN"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"classification-based-grasp-detection-using","title":"Classification based Grasp Detection using Spatial Transformer Network","date":"2018-03-04","arxiv_id":"1803.01356","n_code_links":0,"syntology":null},{"paper":"/paper/deep-neural-network-for-traffic-sign","slug":"deep-neural-network-for-traffic-sign","title":"Deep neural network for traffic sign recognition systems: An analysis of spatial transformers and stochastic optimisation methods","date":"2018-03-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-spatial-transformer-network","title":"Hierarchical Spatial Transformer Network","date":"2018-01-29","arxiv_id":"1801.09467","n_code_links":0,"syntology":null},{"paper":"/paper/see-towards-semi-supervised-end-to-end-scene","slug":"see-towards-semi-supervised-end-to-end-scene","title":"SEE: Towards Semi-Supervised End-to-End Scene Text Recognition","date":"2017-12-14","arxiv_id":"1712.05404","n_code_links":2,"syntology":null},{"paper":"/paper/see-towards-semi-supervisedend-to-end-scene","slug":"see-towards-semi-supervisedend-to-end-scene","title":"SEE: Towards Semi-SupervisedEnd-to-End Scene Text Recognition","date":"2017-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-label-image-recognition-by-recurrently","title":"Multi-label Image Recognition by Recurrently Discovering Attentional Regions","date":"2017-11-08","arxiv_id":"1711.02816","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-patch-matching-using-convolutional","title":"Image Patch Matching Using Convolutional Descriptors with Euclidean Distance","date":"2017-10-31","arxiv_id":"1710.11359","n_code_links":0,"syntology":null},{"paper":"/paper/learning-deep-context-aware-features-over","slug":"learning-deep-context-aware-features-over","title":"Learning Deep Context-aware Features over Body and Latent Parts for Person Re-identification","date":"2017-10-18","arxiv_id":"1710.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-free-form-deformation-network-for-object","title":"Deep Free-Form Deformation Network for Object-Mask Registration","date":"2017-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"recursive-spatial-transformer-rest-for","title":"Recursive Spatial Transformer (ReST) for Alignment-Free Face Recognition","date":"2017-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/3d-morphable-models-as-spatial-transformer","slug":"3d-morphable-models-as-spatial-transformer","title":"3D Morphable Models as Spatial Transformer Networks","date":"2017-08-23","arxiv_id":"1708.07199","n_code_links":1,"syntology":null},{"paper":"/paper/stn-ocr-a-single-neural-network-for-text","slug":"stn-ocr-a-single-neural-network-for-text","title":"STN-OCR: A single Neural Network for Text Detection and Text Recognition","date":"2017-07-27","arxiv_id":"1707.08831","n_code_links":3,"syntology":null},{"paper":null,"slug":"ssemnet-serial-section-electron-microscopy","title":"ssEMnet: Serial-section Electron Microscopy Image Registration using a Spatial Transformer Network with Learned Features","date":"2017-07-25","arxiv_id":"1707.07833","n_code_links":0,"syntology":null},{"paper":"/paper/a-deep-regression-architecture-with-two-stage","slug":"a-deep-regression-architecture-with-two-stage","title":"A Deep Regression Architecture With Two-Stage Re-Initialization for High Performance Facial Landmark Detection","date":"2017-07-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"enriched-deep-recurrent-visual-attention","title":"Enriched Deep Recurrent Visual Attention Model for Multiple Object Recognition","date":"2017-06-12","arxiv_id":"1706.03581","n_code_links":0,"syntology":null},{"paper":null,"slug":"transflow-unsupervised-motion-flow-by-joint","title":"TransFlow: Unsupervised Motion Flow by Joint Geometric and Pixel-level Estimation","date":"2017-06-01","arxiv_id":"1706.00322","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-end-to-end-face-recognition-through","title":"Towards End-to-End Face Recognition through Alignment Learning","date":"2017-01-25","arxiv_id":"1701.07174","n_code_links":0,"syntology":null},{"paper":null,"slug":"areas-of-attention-for-image-captioning","title":"Areas of Attention for Image Captioning","date":"2016-12-03","arxiv_id":"1612.01033","n_code_links":0,"syntology":null},{"paper":"/paper/rmpe-regional-multi-person-pose-estimation","slug":"rmpe-regional-multi-person-pose-estimation","title":"RMPE: Regional Multi-person Pose Estimation","date":"2016-12-01","arxiv_id":"1612.00137","n_code_links":14,"syntology":null},{"paper":null,"slug":"demeshnet-blind-face-inpainting-for-deep","title":"DeMeshNet: Blind Face Inpainting for Deep MeshFace Verification","date":"2016-11-16","arxiv_id":"1611.05271","n_code_links":0,"syntology":null},{"paper":null,"slug":"recurrent-3d-attentional-networks-for-end-to","title":"Recurrent 3D Attentional Networks for End-to-End Active Object Recognition","date":"2016-10-14","arxiv_id":"1610.04308","n_code_links":0,"syntology":null},{"paper":"/paper/star-net-a-spatial-attention-residue-network","slug":"star-net-a-spatial-attention-residue-network","title":"Star-net: A spatial attention residue network for scene text recognition.","date":"2016-09-20","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/gvnn-neural-network-library-for-geometric","slug":"gvnn-neural-network-library-for-geometric","title":"gvnn: Neural Network Library for Geometric Computer Vision","date":"2016-07-25","arxiv_id":"1607.07405","n_code_links":1,"syntology":null},{"paper":null,"slug":"recurrent-attentional-networks-for-saliency","title":"Recurrent Attentional Networks for Saliency Detection","date":"2016-04-12","arxiv_id":"1604.03227","n_code_links":0,"syntology":null},{"paper":"/paper/traffic-sign-classification-using-deep","slug":"traffic-sign-classification-using-deep","title":"Traffic Sign Classification Using Deep Inception Based Convolutional Networks","date":"2015-11-10","arxiv_id":"1511.02992","n_code_links":2,"syntology":null},{"paper":null,"slug":"color-space-transformation-network","title":"Color Space Transformation Network","date":"2015-10-31","arxiv_id":"1511.01064","n_code_links":0,"syntology":null},{"paper":"/paper/recurrent-spatial-transformer-networks","slug":"recurrent-spatial-transformer-networks","title":"Recurrent Spatial Transformer Networks","date":"2015-09-17","arxiv_id":"1509.05329","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/spatial-transformer-networks","slug":"spatial-transformer-networks","title":"Spatial Transformer Networks","date":"2015-06-05","arxiv_id":"1506.02025","n_code_links":45,"syntology":{"ran":21,"of":31,"n_ran_checked":21,"n_instrument":0,"unverified":10,"pointer_only":0,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 1 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","official":null}}],"record_sha256":"cdb50cf83ec9f07538456ba091727aee4be84d6d0d2bab200dda7a46cb32c234","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}