{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/transformer/papers/112","list_of":"/method/transformer","method":"Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":112,"pages_in_order":140,"rows_per_page":100,"rows":[11101,11200],"of":13999,"counts":{"archive_papers_tagged":13999,"with_a_code_link":6572,"where_syntology_ran_a_sample":2248,"not_listed_spam_title":0,"listed":13999,"listed_where_code_ran":2248,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1919,"every_run_a_failure_of_syntologys_instrument":329,"listed_with_a_run_with_no_instrument_failure":1919,"listed_every_run_a_failure_of_syntologys_instrument":329,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/transformer","prev":"/method/transformer/papers/111","next":"/method/transformer/papers/113","papers":[{"paper":"/paper/efficient-visual-tracking-with-exemplar","slug":"efficient-visual-tracking-with-exemplar","title":"Efficient Visual Tracking with Exemplar Transformers","date":"2021-12-17","arxiv_id":"2112.09686","n_code_links":2,"syntology":null},{"paper":"/paper/full-transformer-framework-for-robust-point","slug":"full-transformer-framework-for-robust-point","title":"Full Transformer Framework for Robust Point Cloud Registration with Deep Information Interaction","date":"2021-12-17","arxiv_id":"2112.09385","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporate-dependency-relation-knowledge","title":"Incorporate Dependency Relation Knowledge into Transformer Block for Multi-turn Dialogue Generation","date":"2021-12-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-clarification-question","title":"Self-supervised clarification question generation for ambiguous multi-turn conversation","date":"2021-12-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/towards-end-to-end-image-compression-and","slug":"towards-end-to-end-image-compression-and","title":"Towards End-to-End Image Compression and Analysis with Transformers","date":"2021-12-17","arxiv_id":"2112.09300","n_code_links":1,"syntology":null},{"paper":"/paper/unified-2d-and-3d-pre-training-for-medical","slug":"unified-2d-and-3d-pre-training-for-medical","title":"UniMiSS: Universal Medical Self-Supervised Learning via Breaking Dimensionality Barrier","date":"2021-12-17","arxiv_id":"2112.09356","n_code_links":1,"syntology":null},{"paper":null,"slug":"your-answer-is-incorrect-would-you-like-to","title":"Your Answer is Incorrect... Would you like to know why? Introducing a Bilingual Short Answer Feedback Dataset","date":"2021-12-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/block-skim-efficient-question-answering-for","slug":"block-skim-efficient-question-answering-for","title":"Block-Skim: Efficient Question Answering for Transformer","date":"2021-12-16","arxiv_id":"2112.08560","n_code_links":1,"syntology":null},{"paper":"/paper/dprost-6-dof-object-pose-estimation-using","slug":"dprost-6-dof-object-pose-estimation-using","title":"DProST: Dynamic Projective Spatial Transformer Network for 6D Pose Estimation","date":"2021-12-16","arxiv_id":"2112.08775","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-to-augment-your-vits-consistency-loss-and","title":"How to augment your ViTs? Consistency loss and StyleAug, a random style transfer augmentation","date":"2021-12-16","arxiv_id":"2112.09260","n_code_links":0,"syntology":null},{"paper":"/paper/kat-a-knowledge-augmented-transformer-for","slug":"kat-a-knowledge-augmented-transformer-for","title":"KAT: A Knowledge Augmented Transformer for Vision-and-Language","date":"2021-12-16","arxiv_id":"2112.08614","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-enhanced-session-based","title":"Knowledge-enhanced Session-based Recommendation with Temporal Transformer","date":"2021-12-16","arxiv_id":"2112.08745","n_code_links":0,"syntology":null},{"paper":"/paper/learning-bounded-context-free-grammar-via","slug":"learning-bounded-context-free-grammar-via","title":"Learning Bounded Context-Free-Grammar via LSTM and the Transformer:Difference and Explanations","date":"2021-12-16","arxiv_id":"2112.09174","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shihui2010/learn_cfg_with_neural_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/looking-outside-the-box-to-ground-language-in","slug":"looking-outside-the-box-to-ground-language-in","title":"Bottom Up Top Down Detection Transformers for Language Grounding in Images and Point Clouds","date":"2021-12-16","arxiv_id":"2112.08879","n_code_links":1,"syntology":null},{"paper":null,"slug":"multivariate-realized-volatility-forecasting","title":"Multivariate Realized Volatility Forecasting with Graph Neural Network","date":"2021-12-16","arxiv_id":"2112.09015","n_code_links":0,"syntology":null},{"paper":null,"slug":"sgeitl-scene-graph-enhanced-image-text","title":"SGEITL: Scene Graph Enhanced Image-Text Learning for Visual Commonsense Reasoning","date":"2021-12-16","arxiv_id":"2112.08587","n_code_links":0,"syntology":null},{"paper":"/paper/trading-with-the-momentum-transformer-an","slug":"trading-with-the-momentum-transformer-an","title":"Trading with the Momentum Transformer: An Intelligent and Interpretable Architecture","date":"2021-12-16","arxiv_id":"2112.08534","n_code_links":3,"syntology":null},{"paper":"/paper/transzero-cross-attribute-guided-transformer","slug":"transzero-cross-attribute-guided-transformer","title":"TransZero++: Cross Attribute-Guided Transformer for Zero-Shot Learning","date":"2021-12-16","arxiv_id":"2112.08643","n_code_links":1,"syntology":null},{"paper":null,"slug":"trees-in-transformers-a-theoretical-analysis","title":"Trees in transformers: a theoretical analysis of the Transformer's ability to represent trees","date":"2021-12-16","arxiv_id":"2112.11913","n_code_links":0,"syntology":null},{"paper":"/paper/dense-video-captioning-using-unsupervised","slug":"dense-video-captioning-using-unsupervised","title":"Dense Video Captioning Using Unsupervised Semantic Information","date":"2021-12-15","arxiv_id":"2112.08455","n_code_links":1,"syntology":null},{"paper":"/paper/is-my-favorite-new-movie-my-favorite-movie","slug":"is-my-favorite-new-movie-my-favorite-movie","title":"Is \"My Favorite New Movie\" My Favorite Movie? Probing the Understanding of Recursive Noun Phrases","date":"2021-12-15","arxiv_id":"2112.08326","n_code_links":1,"syntology":null},{"paper":null,"slug":"lesan-machine-translation-for-low-resource","title":"Lesan -- Machine Translation for Low Resource Languages","date":"2021-12-15","arxiv_id":"2112.08191","n_code_links":0,"syntology":null},{"paper":"/paper/oracle-linguistic-graphs-complement-a","slug":"oracle-linguistic-graphs-complement-a","title":"Linguistic Frameworks Go Toe-to-Toe at Neuro-Symbolic Language Modeling","date":"2021-12-15","arxiv_id":"2112.07874","n_code_links":1,"syntology":null},{"paper":"/paper/spts-single-point-text-spotting","slug":"spts-single-point-text-spotting","title":"SPTS: Single-Point Text Spotting","date":"2021-12-15","arxiv_id":"2112.07917","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-based-video-hashing","title":"Vision Transformer Based Video Hashing Retrieval for Tracing the Source of Fake Videos","date":"2021-12-15","arxiv_id":"2112.08117","n_code_links":0,"syntology":null},{"paper":null,"slug":"ace-bert-adversarial-cross-modal-enhanced","title":"ACE-BERT: Adversarial Cross-modal Enhanced BERT for E-commerce Retrieval","date":"2021-12-14","arxiv_id":"2112.07209","n_code_links":0,"syntology":null},{"paper":null,"slug":"epigenomic-language-models-powered-by","title":"Epigenomic language models powered by Cerebras","date":"2021-12-14","arxiv_id":"2112.07571","n_code_links":0,"syntology":null},{"paper":"/paper/geometry-contrastive-transformer-for","slug":"geometry-contrastive-transformer-for","title":"Geometry-Contrastive Transformer for Generalized 3D Pose Transfer","date":"2021-12-14","arxiv_id":"2112.07374","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-hybrid-ctc-attention-end-to-end","title":"Improving Hybrid CTC/Attention End-to-end Speech Recognition with Pretrained Acoustic and Language Model","date":"2021-12-14","arxiv_id":"2112.07254","n_code_links":0,"syntology":null},{"paper":null,"slug":"temporal-transformer-networks-with-self","title":"Temporal Transformer Networks with Self-Supervision for Action Recognition","date":"2021-12-14","arxiv_id":"2112.07338","n_code_links":0,"syntology":null},{"paper":null,"slug":"5th-place-solution-for-vspw-2021-challenge","title":"5th Place Solution for VSPW 2021 Challenge","date":"2021-12-13","arxiv_id":"2112.06379","n_code_links":0,"syntology":null},{"paper":"/paper/dependency-learning-for-legal-judgment","slug":"dependency-learning-for-legal-judgment","title":"Dependency Learning for Legal Judgment Prediction with a Unified Text-to-Text Transformer","date":"2021-12-13","arxiv_id":"2112.06370","n_code_links":1,"syntology":null},{"paper":"/paper/embracing-single-stride-3d-object-detector","slug":"embracing-single-stride-3d-object-detector","title":"Embracing Single Stride 3D Object Detector with Sparse Transformer","date":"2021-12-13","arxiv_id":"2112.06375","n_code_links":2,"syntology":null},{"paper":null,"slug":"hformer-hybrid-cnn-transformer-for-fringe","title":"Hformer: Hybrid CNN-Transformer for Fringe Order Prediction in Phase Unwrapping of Fringe Projection","date":"2021-12-13","arxiv_id":"2112.06759","n_code_links":0,"syntology":null},{"paper":null,"slug":"pedestrian-trajectory-prediction-via-spatial","title":"Pedestrian Trajectory Prediction via Spatial Interaction Transformer Network","date":"2021-12-13","arxiv_id":"2112.06624","n_code_links":0,"syntology":null},{"paper":"/paper/sequential-recommendation-with-bidirectional","slug":"sequential-recommendation-with-bidirectional","title":"Improving Sequential Recommendations via Bidirectional Temporal Data Augmentation with Pre-training","date":"2021-12-13","arxiv_id":"2112.06460","n_code_links":1,"syntology":null},{"paper":"/paper/implicit-transformer-network-for-screen-1","slug":"implicit-transformer-network-for-screen-1","title":"Implicit Transformer Network for Screen Content Image Continuous Super-Resolution","date":"2021-12-12","arxiv_id":"2112.06174","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["codyshen0000/itsrn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-vision-transformers-for-incremental","title":"Improving Vision Transformers for Incremental Learning","date":"2021-12-12","arxiv_id":"2112.06103","n_code_links":0,"syntology":null},{"paper":"/paper/towards-more-efficient-insertion-transformer","slug":"towards-more-efficient-insertion-transformer","title":"Towards More Efficient Insertion Transformer with Fractional Positional Encoding","date":"2021-12-12","arxiv_id":"2112.06295","n_code_links":1,"syntology":null},{"paper":"/paper/composer-compositional-learning-of-group","slug":"composer-compositional-learning-of-group","title":"COMPOSER: Compositional Reasoning of Group Activity in Videos with Keypoint-Only Modality","date":"2021-12-11","arxiv_id":"2112.05892","n_code_links":1,"syntology":null},{"paper":null,"slug":"building-a-great-multi-lingual-teacher-with","title":"Building a great multi-lingual teacher with sparsely-gated mixture of experts for speech recognition","date":"2021-12-10","arxiv_id":"2112.05820","n_code_links":0,"syntology":null},{"paper":"/paper/couplformer-rethinking-vision-transformer","slug":"couplformer-rethinking-vision-transformer","title":"Couplformer:Rethinking Vision Transformer with Coupling Attention Map","date":"2021-12-10","arxiv_id":"2112.05425","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":2,"n_instrument":4,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["wer010/Couplformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/deep-vit-features-as-dense-visual-descriptors","slug":"deep-vit-features-as-dense-visual-descriptors","title":"Deep ViT Features as Dense Visual Descriptors","date":"2021-12-10","arxiv_id":"2112.05814","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/pre-training-and-fine-tuning-transformers-for","slug":"pre-training-and-fine-tuning-transformers-for","title":"Self-Supervised Transformers for fMRI representation","date":"2021-12-10","arxiv_id":"2112.05761","n_code_links":2,"syntology":null},{"paper":null,"slug":"vut-versatile-ui-transformer-for-multi-modal","title":"VUT: Versatile UI Transformer for Multi-Modal Multi-Task User Interface Modeling","date":"2021-12-10","arxiv_id":"2112.05692","n_code_links":0,"syntology":null},{"paper":"/paper/3d-medical-point-transformer-introducing","slug":"3d-medical-point-transformer-introducing","title":"3D Medical Point Transformer: Introducing Convolution to Attention Networks for Medical Point Cloud Analysis","date":"2021-12-09","arxiv_id":"2112.04863","n_code_links":1,"syntology":null},{"paper":"/paper/a-bilingual-openworld-video-text-dataset-and","slug":"a-bilingual-openworld-video-text-dataset-and","title":"A Bilingual, OpenWorld Video Text Dataset and End-to-end Video Text Spotter with Transformer","date":"2021-12-09","arxiv_id":"2112.04888","n_code_links":3,"syntology":{"ran":16,"of":19,"n_ran_checked":14,"n_instrument":2,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 9 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["weijiawu/BOVText-Benchmark","weijiawu/transvtspotter"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"extending-adamw-by-leveraging-its-second","title":"Extending AdamW by Leveraging Its Second Moment and Magnitude","date":"2021-12-09","arxiv_id":"2112.06125","n_code_links":0,"syntology":null},{"paper":"/paper/fast-point-transformer","slug":"fast-point-transformer","title":"Fast Point Transformer","date":"2021-12-09","arxiv_id":"2112.04702","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["POSTECH-CVLab/FastPointTransformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pe-former-pose-estimation-transformer","slug":"pe-former-pose-estimation-transformer","title":"PE-former: Pose Estimation Transformer","date":"2021-12-09","arxiv_id":"2112.04981","n_code_links":1,"syntology":null},{"paper":"/paper/recurrent-glimpse-based-decoder-for-detection","slug":"recurrent-glimpse-based-decoder-for-detection","title":"Recurrent Glimpse-based Decoder for Detection with Transformer","date":"2021-12-09","arxiv_id":"2112.04632","n_code_links":1,"syntology":null},{"paper":"/paper/semi-supervised-medical-image-segmentation-2","slug":"semi-supervised-medical-image-segmentation-2","title":"Semi-Supervised Medical Image Segmentation via Cross Teaching between CNN and Transformer","date":"2021-12-09","arxiv_id":"2112.04894","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HiLab-git/SSL4MIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/garment4d-garment-reconstruction-from-point-1","slug":"garment4d-garment-reconstruction-from-point-1","title":"Garment4D: Garment Reconstruction from Point Cloud Sequences","date":"2021-12-08","arxiv_id":"2112.04159","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hongfz16/garment4d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-language-models-by-retrieving-from","slug":"improving-language-models-by-retrieving-from","title":"Improving language models by retrieving from trillions of tokens","date":"2021-12-08","arxiv_id":"2112.04426","n_code_links":2,"syntology":{"ran":16,"of":23,"n_ran_checked":14,"n_instrument":2,"unverified":7,"pointer_only":3,"phrase":"16 ran (of which 5 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 3 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":null}},{"paper":"/paper/joint-global-and-local-hierarchical-priors","slug":"joint-global-and-local-hierarchical-priors","title":"Joint Global and Local Hierarchical Priors for Learned Image Compression","date":"2021-12-08","arxiv_id":"2112.04487","n_code_links":1,"syntology":null},{"paper":"/paper/staf-a-spatio-temporal-attention-fusion","slug":"staf-a-spatio-temporal-attention-fusion","title":"MASTAF: A Model-Agnostic Spatio-Temporal Attention Fusion Network for Few-shot Video Classification","date":"2021-12-08","arxiv_id":"2112.04585","n_code_links":1,"syntology":null},{"paper":"/paper/transformaly-two-feature-spaces-are-better","slug":"transformaly-two-feature-spaces-are-better","title":"Transformaly -- Two (Feature Spaces) Are Better Than One","date":"2021-12-08","arxiv_id":"2112.04185","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-deep-language-model-to-predict-metabolic","title":"A deep language model to predict metabolic network equilibria","date":"2021-12-07","arxiv_id":"2112.03588","n_code_links":0,"syntology":null},{"paper":"/paper/attention-based-model-and-deep-reinforcement","slug":"attention-based-model-and-deep-reinforcement","title":"Attention-Based Model and Deep Reinforcement Learning for Distribution of Event Processing Tasks","date":"2021-12-07","arxiv_id":"2112.03835","n_code_links":1,"syntology":null},{"paper":"/paper/bootstrapping-vits-towards-liberating-vision","slug":"bootstrapping-vits-towards-liberating-vision","title":"Bootstrapping ViTs: Towards Liberating Vision Transformers from Pre-training","date":"2021-12-07","arxiv_id":"2112.03552","n_code_links":1,"syntology":null},{"paper":"/paper/emulating-spatio-temporal-realizations-of","slug":"emulating-spatio-temporal-realizations-of","title":"Emulating Spatio-Temporal Realizations of Three-Dimensional Isotropic Turbulence via Deep Sequence Learning Models","date":"2021-12-07","arxiv_id":"2112.03469","n_code_links":1,"syntology":null},{"paper":"/paper/regularity-learning-via-explicit-distribution","slug":"regularity-learning-via-explicit-distribution","title":"Regularity Learning via Explicit Distribution Modeling for Skeletal Video Anomaly Detection","date":"2021-12-07","arxiv_id":"2112.03649","n_code_links":1,"syntology":null},{"paper":null,"slug":"relating-transformers-to-models-and-neural-1","title":"Relating transformers to models and neural representations of the hippocampal formation","date":"2021-12-07","arxiv_id":"2112.04035","n_code_links":0,"syntology":null},{"paper":"/paper/ssat-a-symmetric-semantic-aware-transformer","slug":"ssat-a-symmetric-semantic-aware-transformer","title":"SSAT: A Symmetric Semantic-Aware Transformer Network for Makeup Transfer and Removal","date":"2021-12-07","arxiv_id":"2112.03631","n_code_links":2,"syntology":null},{"paper":"/paper/getam-gradient-weighted-element-wise","slug":"getam-gradient-weighted-element-wise","title":"GETAM: Gradient-weighted Element-wise Transformer Attention Map for Weakly-supervised Semantic segmentation","date":"2021-12-06","arxiv_id":"2112.02841","n_code_links":1,"syntology":null},{"paper":"/paper/offline-pre-trained-multi-agent-decision-1","slug":"offline-pre-trained-multi-agent-decision-1","title":"Offline Pre-trained Multi-Agent Decision Transformer: One Big Sequence Model Tackles All SMAC Tasks","date":"2021-12-06","arxiv_id":"2112.02845","n_code_links":1,"syntology":null},{"paper":null,"slug":"one-shot-talking-face-generation-from-single","title":"One-shot Talking Face Generation from Single-speaker Audio-Visual Correlation Learning","date":"2021-12-06","arxiv_id":"2112.02749","n_code_links":0,"syntology":null},{"paper":"/paper/pttr-relational-3d-point-cloud-object","slug":"pttr-relational-3d-point-cloud-object","title":"PTTR: Relational 3D Point Cloud Object Tracking with Transformer","date":"2021-12-06","arxiv_id":"2112.02857","n_code_links":1,"syntology":null},{"paper":"/paper/scaling-up-influence-functions","slug":"scaling-up-influence-functions","title":"Scaling Up Influence Functions","date":"2021-12-06","arxiv_id":"2112.03052","n_code_links":2,"syntology":null},{"paper":null,"slug":"stformer-a-noise-aware-efficient-spatio","title":"Spatio-Temporal meets Wavelet: Disentangled Traffic Flow Forecasting via Efficient Spectral Graph Attention Network","date":"2021-12-06","arxiv_id":"2112.02740","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-token-normalization-improves-vision-1","slug":"dynamic-token-normalization-improves-vision-1","title":"Dynamic Token Normalization Improves Vision Transformers","date":"2021-12-05","arxiv_id":"2112.02624","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["wqshao126/dtn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-tracking-representations-via-dual","slug":"learning-tracking-representations-via-dual","title":"Learning Tracking Representations via Dual-Branch Fully Transformer Networks","date":"2021-12-05","arxiv_id":"2112.02571","n_code_links":1,"syntology":null},{"paper":"/paper/pose-guided-feature-disentangling-for","slug":"pose-guided-feature-disentangling-for","title":"Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer","date":"2021-12-05","arxiv_id":"2112.02466","n_code_links":1,"syntology":null},{"paper":"/paper/3rd-place-a-global-and-local-dual-retrieval","slug":"3rd-place-a-global-and-local-dual-retrieval","title":"3rd Place: A Global and Local Dual Retrieval Solution to Facebook AI Image Similarity Challenge","date":"2021-12-04","arxiv_id":"2112.02373","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-strategy-based-pre-training-method","title":"A Multi-Strategy based Pre-Training Method for Cold-Start Recommendation","date":"2021-12-04","arxiv_id":"2112.02275","n_code_links":0,"syntology":null},{"paper":"/paper/lavt-language-aware-vision-transformer-for","slug":"lavt-language-aware-vision-transformer-for","title":"LAVT: Language-Aware Vision Transformer for Referring Image Segmentation","date":"2021-12-04","arxiv_id":"2112.02244","n_code_links":1,"syntology":null},{"paper":"/paper/u2-former-a-nested-u-shaped-transformer-for","slug":"u2-former-a-nested-u-shaped-transformer-for","title":"U2-Former: A Nested U-shaped Transformer for Image Restoration","date":"2021-12-04","arxiv_id":"2112.02279","n_code_links":0,"syntology":null},{"paper":"/paper/yourtts-towards-zero-shot-multi-speaker-tts","slug":"yourtts-towards-zero-shot-multi-speaker-tts","title":"YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for everyone","date":"2021-12-04","arxiv_id":"2112.02418","n_code_links":3,"syntology":null},{"paper":"/paper/ctin-robust-contextual-transformer-network","slug":"ctin-robust-contextual-transformer-network","title":"CTIN: Robust Contextual Transformer Network for Inertial Navigation","date":"2021-12-03","arxiv_id":"2112.02143","n_code_links":1,"syntology":null},{"paper":"/paper/efficient-two-stage-detection-of-human-object","slug":"efficient-two-stage-detection-of-human-object","title":"Efficient Two-Stage Detection of Human-Object Interactions with a Novel Unary-Pairwise Transformer","date":"2021-12-03","arxiv_id":"2112.01838","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fredzzhang/upt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nn-lut-neural-approximation-of-non-linear","title":"NN-LUT: Neural Approximation of Non-Linear Operations for Efficient Transformer Inference","date":"2021-12-03","arxiv_id":"2112.02191","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-shot-black-box-adversarial-attacks","title":"Single-Shot Black-Box Adversarial Attacks Against Malware Detectors: A Causal Language Model Approach","date":"2021-12-03","arxiv_id":"2112.01724","n_code_links":0,"syntology":null},{"paper":"/paper/transzero-attribute-guided-transformer-for","slug":"transzero-attribute-guided-transformer-for","title":"TransZero: Attribute-guided Transformer for Zero-Shot Learning","date":"2021-12-03","arxiv_id":"2112.01683","n_code_links":1,"syntology":null},{"paper":"/paper/masked-attention-mask-transformer-for","slug":"masked-attention-mask-transformer-for","title":"Masked-attention Mask Transformer for Universal Image Segmentation","date":"2021-12-02","arxiv_id":"2112.01527","n_code_links":7,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/Mask2Former"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/mtfnet-mutual-transformer-fusion-network-for","slug":"mtfnet-mutual-transformer-fusion-network-for","title":"MutualFormer: Multi-Modality Representation Learning via Cross-Diffusion Attention","date":"2021-12-02","arxiv_id":"2112.01177","n_code_links":1,"syntology":null},{"paper":"/paper/plsum-generating-pt-br-wikipedia-by","slug":"plsum-generating-pt-br-wikipedia-by","title":"PLSUM: Generating PT-BR Wikipedia by Summarizing Multiple Websites","date":"2021-12-02","arxiv_id":"2112.01591","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalevlad-improving-multimodal-sentiment","title":"ScaleVLAD: Improving Multimodal Sentiment Analysis via Multi-Scale Fusion of Locally Descriptors","date":"2021-12-02","arxiv_id":"2112.01368","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-video-transformer","slug":"self-supervised-video-transformer","title":"Self-supervised Video Transformer","date":"2021-12-02","arxiv_id":"2112.01514","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["kahnchana/svt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/swintrack-a-simple-and-strong-baseline-for","slug":"swintrack-a-simple-and-strong-baseline-for","title":"SwinTrack: A Simple and Strong Baseline for Transformer Tracking","date":"2021-12-02","arxiv_id":"2112.00995","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["litinglin/swintrack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"tbn-vit-temporal-bilateral-network-with","title":"TBN-ViT: Temporal Bilateral Network with Vision Transformer for Video Scene Parsing","date":"2021-12-02","arxiv_id":"2112.01033","n_code_links":0,"syntology":null},{"paper":"/paper/tctn-a-3d-temporal-convolutional-transformer","slug":"tctn-a-3d-temporal-convolutional-transformer","title":"PTCT: Patches with 3D-Temporal Convolutional Transformer Network for Precipitation Nowcasting","date":"2021-12-02","arxiv_id":"2112.01085","n_code_links":1,"syntology":null},{"paper":"/paper/uni-perceiver-pre-training-unified","slug":"uni-perceiver-pre-training-unified","title":"Uni-Perceiver: Pre-training Unified Architecture for Generic Perception for Zero-shot and Few-shot Tasks","date":"2021-12-02","arxiv_id":"2112.01522","n_code_links":1,"syntology":null},{"paper":null,"slug":"visual-semantic-transformer-for-scene-text","title":"Visual-Semantic Transformer for Scene Text Recognition","date":"2021-12-02","arxiv_id":"2112.00948","n_code_links":0,"syntology":null},{"paper":"/paper/co-evolution-transformer-for-protein-contact","slug":"co-evolution-transformer-for-protein-contact","title":"Co-evolution Transformer for Protein Contact Prediction","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/container-context-aggregation-networks","slug":"container-context-aggregation-networks","title":"Container: Context Aggregation Networks","date":"2021-12-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"cross-view-geo-localization-with-layer-to","title":"Cross-view Geo-localization with Layer-to-Layer Transformer","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"do-transformers-really-perform-badly-for","title":"Do Transformers Really Perform Badly for Graph Representation?","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-split-task-agnostic-vision","title":"Federated Split Task-Agnostic Vision Transformer for COVID-19 CXR Diagnosis","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/focal-attention-for-long-range-interactions","slug":"focal-attention-for-long-range-interactions","title":"Focal Attention for Long-Range Interactions in Vision Transformers","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"gauge-equivariant-transformer","title":"Gauge Equivariant Transformer","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"71d7364dd0eed750e52ff4ec1f98ec275f497a654ad635a5bba4f1fd9af23eb4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}