{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/position-wise-feed-forward-layer/papers/117","list_of":"/method/position-wise-feed-forward-layer","method":"Position-Wise Feed-Forward Layer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":117,"pages_in_order":139,"rows_per_page":100,"rows":[11601,11700],"of":13895,"counts":{"archive_papers_tagged":13895,"with_a_code_link":6514,"where_syntology_ran_a_sample":2229,"not_listed_spam_title":0,"listed":13895,"listed_where_code_ran":2229,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1902,"every_run_a_failure_of_syntologys_instrument":327,"listed_with_a_run_with_no_instrument_failure":1902,"listed_every_run_a_failure_of_syntologys_instrument":327,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/position-wise-feed-forward-layer","prev":"/method/position-wise-feed-forward-layer/papers/116","next":"/method/position-wise-feed-forward-layer/papers/118","papers":[{"paper":"/paper/the-devil-is-in-the-detail-simple-tricks","slug":"the-devil-is-in-the-detail-simple-tricks","title":"The Devil is in the Detail: Simple Tricks Improve Systematic Generalization of Transformers","date":"2021-08-26","arxiv_id":"2108.12284","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["robertcsordas/transformer_generalization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tph-yolov5-improved-yolov5-based-on","slug":"tph-yolov5-improved-yolov5-based-on","title":"TPH-YOLOv5: Improved YOLOv5 Based on Transformer Prediction Head for Object Detection on Drone-captured Scenarios","date":"2021-08-26","arxiv_id":"2108.11539","n_code_links":3,"syntology":null},{"paper":null,"slug":"cancerbert-a-bert-model-for-extracting-breast","title":"CancerBERT: a BERT model for Extracting Breast Cancer Phenotypes from Electronic Health Records","date":"2021-08-25","arxiv_id":"2108.11303","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-transformer-for-single-image-super","slug":"efficient-transformer-for-single-image-super","title":"Transformer for Single Image Super-Resolution","date":"2021-08-25","arxiv_id":"2108.11084","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":7,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luissen/esrt"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"auto-parsing-network-for-image-captioning-and","title":"Auto-Parsing Network for Image Captioning and Visual Question Answering","date":"2021-08-24","arxiv_id":"2108.10568","n_code_links":0,"syntology":null},{"paper":null,"slug":"greenformers-improving-computation-and-memory","title":"Greenformers: Improving Computation and Memory Efficiency in Transformer Models via Low-Rank Approximation","date":"2021-08-24","arxiv_id":"2108.10808","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-bert-encoding-and-sentence-level","title":"Using BERT Encoding and Sentence-Level Language Model for Sentence Ordering","date":"2021-08-24","arxiv_id":"2108.10986","n_code_links":0,"syntology":null},{"paper":"/paper/improving-3d-object-detection-with-channel","slug":"improving-3d-object-detection-with-channel","title":"Improving 3D Object Detection with Channel-wise Transformer","date":"2021-08-23","arxiv_id":"2108.10723","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":5,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["hlsheng1/ct3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/one-tts-alignment-to-rule-them-all","slug":"one-tts-alignment-to-rule-them-all","title":"One TTS Alignment To Rule Them All","date":"2021-08-23","arxiv_id":"2108.10447","n_code_links":3,"syntology":null},{"paper":null,"slug":"recurrent-multiple-shared-layers-in-depth-for","title":"Recurrent multiple shared layers in Depth for Neural Machine Translation","date":"2021-08-23","arxiv_id":"2108.10417","n_code_links":0,"syntology":null},{"paper":"/paper/swinir-image-restoration-using-swin","slug":"swinir-image-restoration-using-swin","title":"SwinIR: Image Restoration Using Swin Transformer","date":"2021-08-23","arxiv_id":"2108.10257","n_code_links":9,"syntology":{"ran":30,"of":45,"n_ran_checked":16,"n_instrument":14,"unverified":15,"pointer_only":5,"phrase":"30 ran (of which 14 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 14 where Syntology's instrument failed) · 15 unverified","official":{"repos":["jingyunliang/swinir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"zs-slr-zero-shot-sign-language-recognition","title":"ZS-SLR: Zero-Shot Sign Language Recognition from RGB-D Videos","date":"2021-08-23","arxiv_id":"2108.10059","n_code_links":0,"syntology":null},{"paper":null,"slug":"guiding-query-position-and-performing-similar","title":"Guiding Query Position and Performing Similar Attention for Transformer-Based Detection Heads","date":"2021-08-22","arxiv_id":"2108.09691","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatial-transformer-networks-for-curriculum","title":"Spatial Transformer Networks for Curriculum Learning","date":"2021-08-22","arxiv_id":"2108.09696","n_code_links":0,"syntology":null},{"paper":null,"slug":"starvqa-space-time-attention-for-video","title":"StarVQA: Space-Time Attention for Video Quality Assessment","date":"2021-08-22","arxiv_id":"2108.09635","n_code_links":0,"syntology":null},{"paper":null,"slug":"uzbert-pretraining-a-bert-model-for-uzbek","title":"UzBERT: pretraining a BERT model for Uzbek","date":"2021-08-22","arxiv_id":"2108.09814","n_code_links":0,"syntology":null},{"paper":null,"slug":"construction-material-classification-on","title":"Construction material classification on imbalanced datasets using Vision Transformer (ViT) architecture","date":"2021-08-21","arxiv_id":"2108.09527","n_code_links":0,"syntology":null},{"paper":"/paper/approximate-bayesian-neural-doppler-imaging","slug":"approximate-bayesian-neural-doppler-imaging","title":"Approximate Bayesian Neural Doppler Imaging","date":"2021-08-20","arxiv_id":"2108.09266","n_code_links":1,"syntology":null},{"paper":null,"slug":"convolutional-neural-network-cnn-vs-visual","title":"Convolutional Neural Network (CNN) vs Vision Transformer (ViT) for Digital Holography","date":"2021-08-20","arxiv_id":"2108.09147","n_code_links":0,"syntology":null},{"paper":"/paper/fastformer-additive-attention-is-all-you-need","slug":"fastformer-additive-attention-is-all-you-need","title":"Fastformer: Additive Attention Can Be All You Need","date":"2021-08-20","arxiv_id":"2108.09084","n_code_links":13,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wuch15/Fastformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/frozen-pretrained-transformers-for-neural","slug":"frozen-pretrained-transformers-for-neural","title":"Frozen Pretrained Transformers for Neural Sign Language Translation","date":"2021-08-20","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"mm-vit-multi-modal-video-transformer-for","title":"MM-ViT: Multi-Modal Video Transformer for Compressed Video Action Recognition","date":"2021-08-20","arxiv_id":"2108.09322","n_code_links":0,"syntology":null},{"paper":"/paper/one-chatbot-per-person-creating-personalized","slug":"one-chatbot-per-person-creating-personalized","title":"One Chatbot Per Person: Creating Personalized Chatbots based on Implicit User Profiles","date":"2021-08-20","arxiv_id":"2108.09355","n_code_links":1,"syntology":null},{"paper":"/paper/pre-training-for-ad-hoc-retrieval-hyperlink","slug":"pre-training-for-ad-hoc-retrieval-hyperlink","title":"Pre-training for Ad-hoc Retrieval: Hyperlink is Also You Need","date":"2021-08-20","arxiv_id":"2108.09346","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-communication-with-adaptive","title":"Semantic Communication with Adaptive Universal Transformer","date":"2021-08-20","arxiv_id":"2108.09119","n_code_links":0,"syntology":null},{"paper":null,"slug":"smart-bird-learnable-sparse-attention-for","title":"Smart Bird: Learnable Sparse Attention for Efficient and Effective Transformer","date":"2021-08-20","arxiv_id":"2108.09193","n_code_links":0,"syntology":null},{"paper":"/paper/trans4trans-efficient-transformer-for-1","slug":"trans4trans-efficient-transformer-for-1","title":"Trans4Trans: Efficient Transformer for Transparent Object and Semantic Scene Segmentation in Real-World Navigation Assistance","date":"2021-08-20","arxiv_id":"2108.09174","n_code_links":1,"syntology":null},{"paper":null,"slug":"uncertainties-and-output-feedback-in-rollout","title":"Uncertainties and output feedback in rollout event-triggered control","date":"2021-08-20","arxiv_id":"2108.09125","n_code_links":0,"syntology":null},{"paper":"/paper/causal-attention-for-unbiased-visual","slug":"causal-attention-for-unbiased-visual","title":"Causal Attention for Unbiased Visual Recognition","date":"2021-08-19","arxiv_id":"2108.08782","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["wangt-cn/caam"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/contrastive-language-image-pre-training-for","slug":"contrastive-language-image-pre-training-for","title":"Contrastive Language-Image Pre-training for the Italian Language","date":"2021-08-19","arxiv_id":"2108.08688","n_code_links":1,"syntology":null},{"paper":"/paper/do-vision-transformers-see-like-convolutional","slug":"do-vision-transformers-see-like-convolutional","title":"Do Vision Transformers See Like Convolutional Neural Networks?","date":"2021-08-19","arxiv_id":"2108.08810","n_code_links":4,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"mvsr-nat-multi-view-subset-regularization-for","title":"MvSR-NAT: Multi-view Subset Regularization for Non-Autoregressive Machine Translation","date":"2021-08-19","arxiv_id":"2108.08447","n_code_links":0,"syntology":null},{"paper":"/paper/video-relation-detection-via-tracklet-based","slug":"video-relation-detection-via-tracklet-based","title":"Video Relation Detection via Tracklet based Visual Transformer","date":"2021-08-19","arxiv_id":"2108.08669","n_code_links":1,"syntology":null},{"paper":"/paper/transformers-predicting-the-future-applying","slug":"transformers-predicting-the-future-applying","title":"Transformers predicting the future. Applying attention in next-frame and time series forecasting","date":"2021-08-18","arxiv_id":"2108.08224","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-streams-and-two-resolution-spectrograms","title":"A Multi-level Acoustic Feature Extraction Framework for Transformer Based End-to-End Speech Recognition","date":"2021-08-18","arxiv_id":"2108.07980","n_code_links":0,"syntology":null},{"paper":"/paper/learning-c-to-x86-translation-an-experiment","slug":"learning-c-to-x86-translation-an-experiment","title":"Learning C to x86 Translation: An Experiment in Neural Compilation","date":"2021-08-17","arxiv_id":"2108.07639","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jordiae/neural-compilers"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/light-field-image-super-resolution-with","slug":"light-field-image-super-resolution-with","title":"Light Field Image Super-Resolution with Transformers","date":"2021-08-17","arxiv_id":"2108.07597","n_code_links":1,"syntology":null},{"paper":null,"slug":"moi-mixer-improving-mlp-mixer-with-multi","title":"MOI-Mixer: Improving MLP-Mixer with Multi Order Interactions in Sequential Recommendation","date":"2021-08-17","arxiv_id":"2108.07505","n_code_links":0,"syntology":null},{"paper":null,"slug":"semantics-aware-attention-improves-neural-1","title":"Semantics-aware Attention Improves Neural Machine Translation","date":"2021-08-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-effective-non-autoregressive-model-for","title":"An Effective Non-Autoregressive Model for Spoken Language Understanding","date":"2021-08-16","arxiv_id":"2108.07005","n_code_links":0,"syntology":null},{"paper":"/paper/no-reference-image-quality-assessment-via-1","slug":"no-reference-image-quality-assessment-via-1","title":"No-Reference Image Quality Assessment via Transformers, Relative Ranking, and Self-Consistency","date":"2021-08-16","arxiv_id":"2108.06858","n_code_links":1,"syntology":null},{"paper":"/paper/scene-designer-a-unified-model-for-scene","slug":"scene-designer-a-unified-model-for-scene","title":"Scene Designer: a Unified Model for Scene Search and Synthesis from Sketch","date":"2021-08-16","arxiv_id":"2108.07353","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-temporal-coherence-for-more-general","slug":"exploring-temporal-coherence-for-more-general","title":"Exploring Temporal Coherence for More General Video Face Forgery Detection","date":"2021-08-15","arxiv_id":"2108.06693","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/sotr-segmenting-objects-with-transformers","slug":"sotr-segmenting-objects-with-transformers","title":"SOTR: Segmenting Objects with Transformers","date":"2021-08-15","arxiv_id":"2108.06747","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["easton-cau/SOTR"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/heterogeneous-temporal-graph-transformer-an","slug":"heterogeneous-temporal-graph-transformer-an","title":"heterogeneous temporal graph transformer: an intelligent system for evolving android malware detection","date":"2021-08-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/conditional-detr-for-fast-training","slug":"conditional-detr-for-fast-training","title":"Conditional DETR for Fast Training Convergence","date":"2021-08-13","arxiv_id":"2108.06152","n_code_links":4,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":4,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["atten4vis/conditionaldetr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/point-voxel-transformer-an-efficient-approach","slug":"point-voxel-transformer-an-efficient-approach","title":"PVT: Point-Voxel Transformer for Point Cloud Learning","date":"2021-08-13","arxiv_id":"2108.06076","n_code_links":2,"syntology":null},{"paper":"/paper/mobile-former-bridging-mobilenet-and","slug":"mobile-former-bridging-mobilenet-and","title":"Mobile-Former: Bridging MobileNet and Transformer","date":"2021-08-12","arxiv_id":"2108.05895","n_code_links":4,"syntology":null},{"paper":"/paper/musiq-multi-scale-image-quality-transformer","slug":"musiq-multi-scale-image-quality-transformer","title":"MUSIQ: Multi-scale Image Quality Transformer","date":"2021-08-12","arxiv_id":"2108.05997","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/tvt-transferable-vision-transformer-for","slug":"tvt-transferable-vision-transformer-for","title":"TVT: Transferable Vision Transformer for Unsupervised Domain Adaptation","date":"2021-08-12","arxiv_id":"2108.05988","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uta-smile/TVT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"abstractive-sentence-summarization-with-1","title":"ICAF: Iterative Contrastive Alignment Framework for Multimodal Abstractive Summarization","date":"2021-08-11","arxiv_id":"2108.05123","n_code_links":0,"syntology":null},{"paper":null,"slug":"convnets-vs-transformers-whose-visual","title":"ConvNets vs. Transformers: Whose Visual Representations are More Transferable?","date":"2021-08-11","arxiv_id":"2108.05305","n_code_links":0,"syntology":null},{"paper":"/paper/perturbing-inputs-for-fragile-interpretations","slug":"perturbing-inputs-for-fragile-interpretations","title":"Perturbing Inputs for Fragile Interpretations in Deep Natural Language Processing","date":"2021-08-11","arxiv_id":"2108.04990","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-multi-resolution-attention-with","title":"Adaptive Multi-Resolution Attention with Linear Complexity","date":"2021-08-10","arxiv_id":"2108.04962","n_code_links":0,"syntology":null},{"paper":"/paper/adarnn-adaptive-learning-and-forecasting-of","slug":"adarnn-adaptive-learning-and-forecasting-of","title":"AdaRNN: Adaptive Learning and Forecasting of Time Series","date":"2021-08-10","arxiv_id":"2108.04443","n_code_links":2,"syntology":null},{"paper":"/paper/differentiable-subset-pruning-of-transformer","slug":"differentiable-subset-pruning-of-transformer","title":"Differentiable Subset Pruning of Transformer Heads","date":"2021-08-10","arxiv_id":"2108.04657","n_code_links":2,"syntology":null},{"paper":null,"slug":"ft-tdr-frequency-guided-transformer-and-top","title":"FT-TDR: Frequency-guided Transformer and Top-Down Refinement Network for Blind Face Inpainting","date":"2021-08-10","arxiv_id":"2108.04424","n_code_links":0,"syntology":null},{"paper":null,"slug":"intent5-search-result-diversification-using","title":"IntenT5: Search Result Diversification using Causal Language Models","date":"2021-08-09","arxiv_id":"2108.04026","n_code_links":0,"syntology":null},{"paper":"/paper/making-transformers-solve-compositional-tasks","slug":"making-transformers-solve-compositional-tasks","title":"Making Transformers Solve Compositional Tasks","date":"2021-08-09","arxiv_id":"2108.04378","n_code_links":1,"syntology":null},{"paper":"/paper/paint-transformer-feed-forward-neural","slug":"paint-transformer-feed-forward-neural","title":"Paint Transformer: Feed Forward Neural Painting with Stroke Prediction","date":"2021-08-09","arxiv_id":"2108.03798","n_code_links":2,"syntology":null},{"paper":"/paper/raftmlp-do-mlp-based-models-dream-of-winning","slug":"raftmlp-do-mlp-based-models-dream-of-winning","title":"RaftMLP: How Much Can Be Done Without Attention and with Less Spatial Locality?","date":"2021-08-09","arxiv_id":"2108.04384","n_code_links":2,"syntology":null},{"paper":null,"slug":"the-hw-tsc-s-offline-speech-translation","title":"The HW-TSC's Offline Speech Translation Systems for IWSLT 2021 Evaluation","date":"2021-08-09","arxiv_id":"2108.03845","n_code_links":0,"syntology":null},{"paper":"/paper/edge-augmented-graph-transformers-global-self","slug":"edge-augmented-graph-transformers-global-self","title":"Global Self-Attention as a Replacement for Graph Convolution","date":"2021-08-07","arxiv_id":"2108.03348","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shamim-hussain/egt","shamim-hussain/egt_pytorch","shamim-hussain/egt_triangular"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hetemotionnet-two-stream-heterogeneous-graph","slug":"hetemotionnet-two-stream-heterogeneous-graph","title":"HetEmotionNet: Two-Stream Heterogeneous Graph Recurrent Neural Network for Multi-modal Emotion Recognition","date":"2021-08-07","arxiv_id":"2108.03354","n_code_links":2,"syntology":null},{"paper":"/paper/rethinking-of-alphastar","slug":"rethinking-of-alphastar","title":"Rethinking of AlphaStar","date":"2021-08-07","arxiv_id":"2108.03452","n_code_links":2,"syntology":null},{"paper":null,"slug":"vision-transformers-for-femur-fracture","title":"Vision Transformer for femur fracture classification","date":"2021-08-07","arxiv_id":"2108.03414","n_code_links":0,"syntology":null},{"paper":null,"slug":"rollout-event-triggered-control-reconciling","title":"Rollout event-triggered control: reconciling event- and time-triggered control","date":"2021-08-06","arxiv_id":"2108.02994","n_code_links":0,"syntology":null},{"paper":"/paper/simpler-is-better-few-shot-semantic","slug":"simpler-is-better-few-shot-semantic","title":"Simpler is Better: Few-shot Semantic Segmentation with Classifier Weight Transformer","date":"2021-08-06","arxiv_id":"2108.03032","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhiheLu/CWT-for-FSS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-right-to-talk-an-audio-visual-transformer","slug":"the-right-to-talk-an-audio-visual-transformer","title":"The Right to Talk: An Audio-Visual Transformer Approach","date":"2021-08-06","arxiv_id":"2108.03256","n_code_links":1,"syntology":null},{"paper":"/paper/fast-convergence-of-detr-with-spatially-1","slug":"fast-convergence-of-detr-with-spatially-1","title":"Fast Convergence of DETR with Spatially Modulated Co-Attention","date":"2021-08-05","arxiv_id":"2108.02404","n_code_links":1,"syntology":null},{"paper":"/paper/finetuning-pretrained-transformers-into","slug":"finetuning-pretrained-transformers-into","title":"Finetuning Pretrained Transformers into Variational Autoencoders","date":"2021-08-05","arxiv_id":"2108.02446","n_code_links":1,"syntology":null},{"paper":null,"slug":"transrefer3d-entity-and-relation-aware","title":"TransRefer3D: Entity-and-Relation Aware Transformer for Fine-Grained 3D Visual Grounding","date":"2021-08-05","arxiv_id":"2108.02388","n_code_links":0,"syntology":null},{"paper":null,"slug":"wechat-neural-machine-translation-systems-for-1","title":"WeChat Neural Machine Translation Systems for WMT21","date":"2021-08-05","arxiv_id":"2108.02401","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-latency-for-online-video","title":"Optimizing Latency for Online Video CaptioningUsing Audio-Visual Transformers","date":"2021-08-04","arxiv_id":"2108.02147","n_code_links":0,"syntology":null},{"paper":null,"slug":"random-offset-block-embedding-array-robe-for","title":"Random Offset Block Embedding Array (ROBE) for CriteoTB Benchmark MLPerf DLRM Model : 1000$\\times$ Compression and 3.1$\\times$ Faster Inference","date":"2021-08-04","arxiv_id":"2108.02191","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-dynamic-head-importance-computation","title":"A Dynamic Head Importance Computation Mechanism for Neural Machine Translation","date":"2021-08-03","arxiv_id":"2108.01377","n_code_links":0,"syntology":null},{"paper":"/paper/a-study-of-multilingual-end-to-end-speech","slug":"a-study-of-multilingual-end-to-end-speech","title":"A Study of Multilingual End-to-End Speech Recognition for Kazakh, Russian, and English","date":"2021-08-03","arxiv_id":"2108.01280","n_code_links":1,"syntology":null},{"paper":"/paper/vision-transformer-with-progressive-sampling","slug":"vision-transformer-with-progressive-sampling","title":"Vision Transformer with Progressive Sampling","date":"2021-08-03","arxiv_id":"2108.01684","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yuexy/PS-ViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/congested-crowd-instance-localization-with","slug":"congested-crowd-instance-localization-with","title":"Congested Crowd Instance Localization with Dilated Convolutional Swin Transformer","date":"2021-08-02","arxiv_id":"2108.00584","n_code_links":1,"syntology":null},{"paper":"/paper/constrained-graphic-layout-generation-via","slug":"constrained-graphic-layout-generation-via","title":"Constrained Graphic Layout Generation via Latent Optimization","date":"2021-08-02","arxiv_id":"2108.00871","n_code_links":1,"syntology":null},{"paper":null,"slug":"musical-speech-a-transformer-based","title":"Musical Speech: A Transformer-based Composition Tool","date":"2021-08-02","arxiv_id":"2108.01043","n_code_links":0,"syntology":null},{"paper":"/paper/representation-learning-for-neural-population","slug":"representation-learning-for-neural-population","title":"Representation learning for neural population activity with Neural Data Transformers","date":"2021-08-02","arxiv_id":"2108.01210","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["snel-repo/neural-data-transformers"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-supervised-answer-retrieval-on-clinical","title":"Self-supervised Answer Retrieval on Clinical Notes","date":"2021-08-02","arxiv_id":"2108.00775","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bidirectional-transformer-based-alignment","title":"A Bidirectional Transformer Based Alignment Model for Unsupervised Word Alignment","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/are-pretrained-convolutions-better-than","slug":"are-pretrained-convolutions-better-than","title":"Are Pretrained Convolutions Better than Pretrained Transformers?","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/beyond-sentence-level-end-to-end-speech","slug":"beyond-sentence-level-end-to-end-speech","title":"Beyond Sentence-Level End-to-End Speech Translation: Context Helps","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/ctfn-hierarchical-learning-for-multimodal","slug":"ctfn-hierarchical-learning-for-multimodal","title":"CTFN: Hierarchical Learning for Multimodal Sentiment Analysis Using Coupled-Translation Fusion Network","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-differential-amplifier-for-extractive","title":"Deep Differential Amplifier for Extractive Summarization","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/ecco-an-open-source-library-for-the","slug":"ecco-an-open-source-library-for-the","title":"Ecco: An Open Source Library for the Explainability of Transformer Language Models","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"fast-and-accurate-neural-machine-translation","title":"Fast and Accurate Neural Machine Translation with Translation Memory","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/lasor-learning-accurate-3d-human-pose-and","slug":"lasor-learning-accurate-3d-human-pose-and","title":"LASOR: Learning Accurate 3D Human Pose and Shape Via Synthetic Occlusion-Aware Data and Neural Mesh Rendering","date":"2021-08-01","arxiv_id":"2108.00351","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["iGame-Lab/LASOR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"modeling-task-aware-mimo-cardinality-for","title":"Modeling Task-Aware MIMO Cardinality for Efficient Multilingual Neural Machine Translation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/more-than-text-multi-modal-chinese-word","slug":"more-than-text-multi-modal-chinese-word","title":"More than Text: Multi-modal Chinese Word Segmentation","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-head-highly-parallelized-lstm-decoder","title":"Multi-Head Highly Parallelized LSTM Decoder for Neural Machine Translation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"point-disambiguate-and-copy-incorporating","title":"Point, Disambiguate and Copy: Incorporating Bilingual Dictionaries for Neural Machine Translation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"stretch-vst-getting-flexible-with-visual","title":"Stretch-VST: Getting Flexible With Visual Stories","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"synchronous-syntactic-attention-for","title":"Synchronous Syntactic Attention for Transformer Neural Machine Translation","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"to-pos-tag-or-not-to-pos-tag-the-impact-of","title":"To POS Tag or Not to POS Tag: The Impact of POS Tags on Morphological Learning in Low-Resource Settings","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-deep-imitation-learning-for","title":"Transformer-based deep imitation learning for dual-arm robot manipulation","date":"2021-08-01","arxiv_id":"2108.00385","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-map-matching-with-model","title":"Transformer-based Map Matching Model with Limited Ground-Truth Data using Transfer-Learning Approach","date":"2021-08-01","arxiv_id":"2108.00439","n_code_links":0,"syntology":null}],"record_sha256":"e4165e28ccf162a444d434c6be2f4669dee474b857acdf6783540784fa2392d7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}