{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/transformer/papers/117","list_of":"/method/transformer","method":"Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":117,"pages_in_order":140,"rows_per_page":100,"rows":[11601,11700],"of":13999,"counts":{"archive_papers_tagged":13999,"with_a_code_link":6572,"where_syntology_ran_a_sample":2248,"not_listed_spam_title":0,"listed":13999,"listed_where_code_ran":2248,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1919,"every_run_a_failure_of_syntologys_instrument":329,"listed_with_a_run_with_no_instrument_failure":1919,"listed_every_run_a_failure_of_syntologys_instrument":329,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/transformer","prev":"/method/transformer/papers/116","next":"/method/transformer/papers/118","papers":[{"paper":"/paper/long-range-transformers-for-dynamic","slug":"long-range-transformers-for-dynamic","title":"Long-Range Transformers for Dynamic Spatiotemporal Forecasting","date":"2021-09-24","arxiv_id":"2109.12218","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["qdata/spacetimeformer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/transformers-generalize-linearly","slug":"transformers-generalize-linearly","title":"Transformers Generalize Linearly","date":"2021-09-24","arxiv_id":"2109.12036","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-linguistic-knowledge-for","title":"Dependency Structure for News Document Summarization","date":"2021-09-23","arxiv_id":"2109.11199","n_code_links":0,"syntology":null},{"paper":null,"slug":"oh-former-omni-relational-high-order","title":"OH-Former: Omni-Relational High-Order Transformer for Person Re-Identification","date":"2021-09-23","arxiv_id":"2109.11159","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-volctrans-glat-system-non-autoregressive","title":"The Volctrans GLAT System: Non-autoregressive Translation Meets WMT21","date":"2021-09-23","arxiv_id":"2109.11247","n_code_links":0,"syntology":null},{"paper":"/paper/controlled-evaluation-of-grammatical","slug":"controlled-evaluation-of-grammatical","title":"Controlled Evaluation of Grammatical Knowledge in Mandarin Chinese Language Models","date":"2021-09-22","arxiv_id":"2109.11058","n_code_links":1,"syntology":null},{"paper":"/paper/hierarchical-multimodal-transformer-to","slug":"hierarchical-multimodal-transformer-to","title":"Hierarchical Multimodal Transformer to Summarize Videos","date":"2021-09-22","arxiv_id":"2109.10559","n_code_links":0,"syntology":null},{"paper":"/paper/kd-vlp-improving-end-to-end-vision-and","slug":"kd-vlp-improving-end-to-end-vision-and","title":"KD-VLP: Improving End-to-End Vision-and-Language Pretraining with Object Knowledge Distillation","date":"2021-09-22","arxiv_id":"2109.10504","n_code_links":1,"syntology":null},{"paper":"/paper/scale-efficiently-insights-from-pre-training","slug":"scale-efficiently-insights-from-pre-training","title":"Scale Efficiently: Insights from Pre-training and Fine-tuning Transformers","date":"2021-09-22","arxiv_id":"2109.10686","n_code_links":3,"syntology":{"ran":12,"of":12,"n_ran_checked":12,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 4 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"t6d-direct-transformers-for-multi-object-6d","title":"T6D-Direct: Transformers for Multi-Object 6D Pose Direct Regression","date":"2021-09-22","arxiv_id":"2109.10948","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-niutrans-machine-translation-systems-for-1","title":"The NiuTrans Machine Translation Systems for WMT21","date":"2021-09-22","arxiv_id":"2109.10485","n_code_links":0,"syntology":null},{"paper":"/paper/ds-net-dynamic-weight-slicing-for-efficient","slug":"ds-net-dynamic-weight-slicing-for-efficient","title":"DS-Net++: Dynamic Weight Slicing for Efficient Inference in CNNs and Transformers","date":"2021-09-21","arxiv_id":"2109.10060","n_code_links":1,"syntology":null},{"paper":"/paper/lotr-face-landmark-localization-using","slug":"lotr-face-landmark-localization-using","title":"LOTR: Face Landmark Localization Using Localization Transformer","date":"2021-09-21","arxiv_id":"2109.10057","n_code_links":0,"syntology":null},{"paper":"/paper/trocr-transformer-based-optical-character","slug":"trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","n_code_links":8,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"dyadformer-a-multi-modal-transformer-for-long","title":"Dyadformer: A Multi-modal Transformer for Long-Range Modeling of Dyadic Interactions","date":"2021-09-20","arxiv_id":"2109.09487","n_code_links":0,"syntology":null},{"paper":"/paper/well-googled-is-half-done-multimodal","slug":"well-googled-is-half-done-multimodal","title":"Well Googled is Half Done: Multimodal Forecasting of New Fashion Product Sales with Image-based Google Trends","date":"2021-09-20","arxiv_id":"2109.09824","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["humaticslab/gtm-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"do-long-range-language-models-actually-use","title":"Do Long-Range Language Models Actually Use Long-Range Context?","date":"2021-09-19","arxiv_id":"2109.09115","n_code_links":0,"syntology":null},{"paper":"/paper/the-seismo-performer-a-novel-machine-learning","slug":"the-seismo-performer-a-novel-machine-learning","title":"The Seismo-Performer: A Novel Machine Learning Approach for General and Efficient Seismic Phase Recognition from Local Earthquakes in Real Time","date":"2021-09-19","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/efficient-hybrid-transformer-learning-global","slug":"efficient-hybrid-transformer-learning-global","title":"UNetFormer: A UNet-like Transformer for Efficient Semantic Segmentation of Remote Sensing Urban Scene Imagery","date":"2021-09-18","arxiv_id":"2109.08937","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WangLibo1995/GeoSeg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sdtp-semantic-aware-decoupled-transformer","title":"SDTP: Semantic-aware Decoupled Transformer Pyramid for Dense Image Prediction","date":"2021-09-18","arxiv_id":"2109.08963","n_code_links":0,"syntology":null},{"paper":"/paper/towards-high-quality-temporal-action","slug":"towards-high-quality-temporal-action","title":"Towards High-Quality Temporal Action Detection with Sparse Proposals","date":"2021-09-18","arxiv_id":"2109.08847","n_code_links":1,"syntology":null},{"paper":null,"slug":"continuous-streaming-multi-talker-asr-with","title":"Continuous Streaming Multi-Talker ASR with Dual-path Transducers","date":"2021-09-17","arxiv_id":"2109.08555","n_code_links":0,"syntology":null},{"paper":null,"slug":"digging-errors-in-nmt-evaluating-and","title":"Digging Errors in NMT: Evaluating and Understanding Model Errors from Hypothesis Distribution","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"expression-snippet-transformer-for-robust","title":"Expression Snippet Transformer for Robust Video-based Facial Expression Recognition","date":"2021-09-17","arxiv_id":"2109.08409","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-known-to-unknown-knowledge-guided","title":"From Known to Unknown: Knowledge-guided Transformer for Time-Series Sales Forecasting in Alibaba","date":"2021-09-17","arxiv_id":"2109.08381","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-low-frequency-patterns-with-a-pre","title":"Learning Low-frequency Patterns with A Pre-trained Document-Grounded Conversation Model","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/primer-searching-for-efficient-transformers","slug":"primer-searching-for-efficient-transformers","title":"Primer: Searching for Efficient Transformers for Language Modeling","date":"2021-09-17","arxiv_id":"2109.08668","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"scaling-laws-vs-model-architectures-how-does","title":"Scaling Laws vs Model Architectures: How does Inductive Bias Influence Scaling? An Extensive Empirical Study on Language Tasks","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-jhu-microsoft-submission-for-wmt21","title":"The JHU-Microsoft Submission for WMT21 Quality Estimation Shared Task","date":"2021-09-17","arxiv_id":"2109.08724","n_code_links":0,"syntology":null},{"paper":"/paper/an-end-to-end-transformer-model-for-3d-object","slug":"an-end-to-end-transformer-model-for-3d-object","title":"An End-to-End Transformer Model for 3D Object Detection","date":"2021-09-16","arxiv_id":"2109.08141","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/fast-slow-transformer-for-visually-grounding","slug":"fast-slow-transformer-for-visually-grounding","title":"Fast-Slow Transformer for Visually Grounding Speech","date":"2021-09-16","arxiv_id":"2109.08186","n_code_links":1,"syntology":null},{"paper":"/paper/label-attention-transformer-with","slug":"label-attention-transformer-with","title":"Label-Attention Transformer with Geometrically Coherent Objects for Image Captioning","date":"2021-09-16","arxiv_id":"2109.07799","n_code_links":1,"syntology":null},{"paper":"/paper/melt-message-level-transformer-with-masked","slug":"melt-message-level-transformer-with-masked","title":"MeLT: Message-Level Transformer with Masked Document Representations as Pre-Training for Stance Detection","date":"2021-09-16","arxiv_id":"2109.08113","n_code_links":1,"syntology":null},{"paper":null,"slug":"scaling-laws-for-neural-machine-translation","title":"Scaling Laws for Neural Machine Translation","date":"2021-09-16","arxiv_id":"2109.07740","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-factorization-of-large-square-matrices","slug":"sparse-factorization-of-large-square-matrices","title":"Sparse Factorization of Large Square Matrices","date":"2021-09-16","arxiv_id":"2109.08184","n_code_links":1,"syntology":null},{"paper":null,"slug":"tanet-a-new-paradigm-for-global-face-super","title":"TANet: A new Paradigm for Global Face Super-resolution via Transformer-CNN Aggregation Network","date":"2021-09-16","arxiv_id":"2109.08174","n_code_links":0,"syntology":null},{"paper":"/paper/the-niutrans-system-for-the-wmt21-efficiency","slug":"the-niutrans-system-for-the-wmt21-efficiency","title":"The NiuTrans System for the WMT21 Efficiency Task","date":"2021-09-16","arxiv_id":"2109.08003","n_code_links":1,"syntology":null},{"paper":"/paper/the-niutrans-system-for-wngt-2020-efficiency-1","slug":"the-niutrans-system-for-wngt-2020-efficiency-1","title":"The NiuTrans System for WNGT 2020 Efficiency Task","date":"2021-09-16","arxiv_id":"2109.08008","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["NiuTrans/NiuTrans.NMT","niutrans/niutensor"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"utterance-level-neural-confidence-measure-for","title":"Utterance-level neural confidence measure for end-to-end children speech recognition","date":"2021-09-16","arxiv_id":"2109.07750","n_code_links":0,"syntology":null},{"paper":"/paper/anchor-detr-query-design-for-transformer","slug":"anchor-detr-query-design-for-transformer","title":"Anchor DETR: Query Design for Transformer-Based Object Detection","date":"2021-09-15","arxiv_id":"2109.07107","n_code_links":2,"syntology":null},{"paper":"/paper/hybrid-local-global-transformer-for-image","slug":"hybrid-local-global-transformer-for-image","title":"Complementary Feature Enhanced Network with Vision Transformer for Image Dehazing","date":"2021-09-15","arxiv_id":"2109.07100","n_code_links":1,"syntology":null},{"paper":"/paper/incorporating-residual-and-normalization","slug":"incorporating-residual-and-normalization","title":"Incorporating Residual and Normalization Layers into Analysis of Masked Language Models","date":"2021-09-15","arxiv_id":"2109.07152","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gorokoba560/norm-analysis-of-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/missformer-an-effective-medical-image","slug":"missformer-an-effective-medical-image","title":"MISSFormer: An Effective Medical Image Segmentation Transformer","date":"2021-09-15","arxiv_id":"2109.07162","n_code_links":1,"syntology":null},{"paper":"/paper/pnp-detr-towards-efficient-visual-analysis","slug":"pnp-detr-towards-efficient-visual-analysis","title":"PnP-DETR: Towards Efficient Visual Analysis with Transformers","date":"2021-09-15","arxiv_id":"2109.07036","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["twangnh/pnp-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pose-transformers-potr-human-motion","slug":"pose-transformers-potr-human-motion","title":"Pose Transformers (POTR): Human Motion Prediction with Non-Autoregressive Transformers","date":"2021-09-15","arxiv_id":"2109.07531","n_code_links":1,"syntology":null},{"paper":"/paper/retroprime-a-diverse-plausible-and","slug":"retroprime-a-diverse-plausible-and","title":"RetroPrime: A Diverse, plausible and Transformer-based method for Single-Step retrosynthesis predictions","date":"2021-09-15","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/sequence-length-is-a-domain-length-based","slug":"sequence-length-is-a-domain-length-based","title":"Sequence Length is a Domain: Length-based Overfitting in Transformer Models","date":"2021-09-15","arxiv_id":"2109.07276","n_code_links":1,"syntology":null},{"paper":"/paper/supcl-seq-supervised-contrastive-learning-for","slug":"supcl-seq-supervised-contrastive-learning-for","title":"SupCL-Seq: Supervised Contrastive Learning for Downstream Optimized Sequence Representations","date":"2021-09-15","arxiv_id":"2109.07424","n_code_links":1,"syntology":null},{"paper":"/paper/towards-incremental-transformers-an-empirical","slug":"towards-incremental-transformers-an-empirical","title":"Towards Incremental Transformers: An Empirical Analysis of Transformer Models for Incremental NLU","date":"2021-09-15","arxiv_id":"2109.07364","n_code_links":1,"syntology":null},{"paper":"/paper/transformer-based-lexically-constrained","slug":"transformer-based-lexically-constrained","title":"Transformer-based Lexically Constrained Headline Generation","date":"2021-09-15","arxiv_id":"2109.07080","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-three-step-training-approach-with-data","title":"A Three Step Training Approach with Data Augmentation for Morphological Inflection","date":"2021-09-14","arxiv_id":"2109.07006","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-biomedical-bert-models-for","title":"Evaluating Biomedical BERT Models for Vocabulary Alignment at Scale in the UMLS Metathesaurus","date":"2021-09-14","arxiv_id":"2109.13348","n_code_links":0,"syntology":null},{"paper":"/paper/structure-enhanced-pop-music-generation-via","slug":"structure-enhanced-pop-music-generation-via","title":"Structure-Enhanced Pop Music Generation via Harmony-Aware Learning","date":"2021-09-14","arxiv_id":"2109.06441","n_code_links":1,"syntology":null},{"paper":null,"slug":"vision-transformer-for-learning-driving","title":"Vision Transformer for Learning Driving Policies in Complex Multi-Agent Environments","date":"2021-09-14","arxiv_id":"2109.06514","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-weights-in-transformer-nmt-fail","title":"Attention Weights in Transformer NMT Fail Aligning Words Between Sequences but Largely Explain Model Predictions","date":"2021-09-13","arxiv_id":"2109.05853","n_code_links":0,"syntology":null},{"paper":"/paper/cdtrans-cross-domain-transformer-for","slug":"cdtrans-cross-domain-transformer-for","title":"CDTrans: Cross-domain Transformer for Unsupervised Domain Adaptation","date":"2021-09-13","arxiv_id":"2109.06165","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cdtrans/cdtrans"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cpt-a-pre-trained-unbalanced-transformerfor","slug":"cpt-a-pre-trained-unbalanced-transformerfor","title":"CPT: A Pre-Trained Unbalanced Transformer for Both Chinese Language Understanding and Generation","date":"2021-09-13","arxiv_id":"2109.05729","n_code_links":1,"syntology":null},{"paper":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-pursuit-of-designing-multi-modal","title":"On Pursuit of Designing Multi-modal Transformer for Video Grounding","date":"2021-09-13","arxiv_id":"2109.06085","n_code_links":0,"syntology":null},{"paper":"/paper/pack-together-entity-and-relation-extraction","slug":"pack-together-entity-and-relation-extraction","title":"Packed Levitated Marker for Entity and Relation Extraction","date":"2021-09-13","arxiv_id":"2109.06067","n_code_links":2,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["thunlp/pl-marker"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/artiboost-boosting-articulated-3d-hand-object","slug":"artiboost-boosting-articulated-3d-hand-object","title":"ArtiBoost: Boosting Articulated 3D Hand-Object Pose Estimation via Online Exploration and Synthesis","date":"2021-09-12","arxiv_id":"2109.05488","n_code_links":2,"syntology":null},{"paper":null,"slug":"constructing-phrase-level-semantic-labels-to","title":"Constructing Phrase-level Semantic Labels to Form Multi-Grained Supervision for Image-Text Retrieval","date":"2021-09-12","arxiv_id":"2109.05523","n_code_links":0,"syntology":null},{"paper":"/paper/levenshtein-training-for-word-level-quality","slug":"levenshtein-training-for-word-level-quality","title":"Levenshtein Training for Word-level Quality Estimation","date":"2021-09-12","arxiv_id":"2109.05611","n_code_links":1,"syntology":null},{"paper":null,"slug":"single-read-reconstruction-for-dna-data","title":"Single-Read Reconstruction for DNA Data Storage Using Transformers","date":"2021-09-12","arxiv_id":"2109.05478","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-mlp-for-image-recognition-is-self","slug":"sparse-mlp-for-image-recognition-is-self","title":"Sparse MLP for Image Recognition: Is Self-Attention Really Necessary?","date":"2021-09-12","arxiv_id":"2109.05422","n_code_links":2,"syntology":null},{"paper":"/paper/teasel-a-transformer-based-speech-prefixed","slug":"teasel-a-transformer-based-speech-prefixed","title":"TEASEL: A Transformer-Based Speech-Prefixed Language Model","date":"2021-09-12","arxiv_id":"2109.05522","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"bornon-bengali-image-captioning-with","title":"Bornon: Bengali Image Captioning with Transformer-based Deep learning approach","date":"2021-09-11","arxiv_id":"2109.05218","n_code_links":0,"syntology":null},{"paper":"/paper/empirical-analysis-of-training-strategies-of","slug":"empirical-analysis-of-training-strategies-of","title":"Empirical Analysis of Training Strategies of Transformer-based Japanese Chit-chat Systems","date":"2021-09-11","arxiv_id":"2109.05217","n_code_links":1,"syntology":null},{"paper":"/paper/multilingual-translation-via-grafting-pre","slug":"multilingual-translation-via-grafting-pre","title":"Multilingual Translation via Grafting Pre-trained Language Models","date":"2021-09-11","arxiv_id":"2109.05256","n_code_links":1,"syntology":null},{"paper":null,"slug":"real-time-multimodal-image-registration-with","title":"Real-time multimodal image registration with partial intraoperative point-set data","date":"2021-09-10","arxiv_id":"2109.05023","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-pyramid-transformer-with-multimodal","slug":"temporal-pyramid-transformer-with-multimodal","title":"Temporal Pyramid Transformer with Multimodal Interaction for Video Question Answering","date":"2021-09-10","arxiv_id":"2109.04735","n_code_links":1,"syntology":null},{"paper":"/paper/a-three-stage-learning-framework-for-low","slug":"a-three-stage-learning-framework-for-low","title":"A Three-Stage Learning Framework for Low-Resource Knowledge-Grounded Dialogue Generation","date":"2021-09-09","arxiv_id":"2109.04096","n_code_links":1,"syntology":{"ran":18,"of":27,"n_ran_checked":12,"n_instrument":6,"unverified":9,"pointer_only":5,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","official":{"repos":["neukg/kat-tslf"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/bag-of-tricks-for-optimizing-transformer","slug":"bag-of-tricks-for-optimizing-transformer","title":"Bag of Tricks for Optimizing Transformer Efficiency","date":"2021-09-09","arxiv_id":"2109.04030","n_code_links":1,"syntology":null},{"paper":null,"slug":"dan-decentralized-attention-based-neural","title":"DAN: Decentralized Attention-based Neural Network for the MinMax Multiple Traveling Salesman Problem","date":"2021-09-09","arxiv_id":"2109.04205","n_code_links":0,"syntology":null},{"paper":"/paper/esimcse-enhanced-sample-building-method-for","slug":"esimcse-enhanced-sample-building-method-for","title":"ESimCSE: Enhanced Sample Building Method for Contrastive Learning of Unsupervised Sentence Embedding","date":"2021-09-09","arxiv_id":"2109.04380","n_code_links":2,"syntology":null},{"paper":null,"slug":"graph-based-decoding-for-task-oriented","title":"Graph-Based Decoding for Task Oriented Semantic Parsing","date":"2021-09-09","arxiv_id":"2109.04587","n_code_links":0,"syntology":null},{"paper":"/paper/mate-multi-view-attention-for-table","slug":"mate-multi-view-attention-for-table","title":"MATE: Multi-view Attention for Table Transformer Efficiency","date":"2021-09-09","arxiv_id":"2109.04312","n_code_links":1,"syntology":null},{"paper":"/paper/thinking-clearly-talking-fast-concept-guided","slug":"thinking-clearly-talking-fast-concept-guided","title":"Thinking Clearly, Talking Fast: Concept-Guided Non-Autoregressive Generation for Open-Domain Dialogue Systems","date":"2021-09-09","arxiv_id":"2109.04084","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["rowitzou/cg-nar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/uctransnet-rethinking-the-skip-connections-in","slug":"uctransnet-rethinking-the-skip-connections-in","title":"UCTransNet: Rethinking the Skip Connections in U-Net from a Channel-wise Perspective with Transformer","date":"2021-09-09","arxiv_id":"2109.04335","n_code_links":3,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mcgregorwwww/uctransnet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/variational-latent-state-gpt-for-semi","slug":"variational-latent-state-gpt-for-semi","title":"Variational Latent-State GPT for Semi-Supervised Task-Oriented Dialog Systems","date":"2021-09-09","arxiv_id":"2109.04314","n_code_links":2,"syntology":null},{"paper":"/paper/label-verbalization-and-entailment-for","slug":"label-verbalization-and-entailment-for","title":"Label Verbalization and Entailment for Effective Zero- and Few-Shot Relation Extraction","date":"2021-09-08","arxiv_id":"2109.03659","n_code_links":1,"syntology":null},{"paper":"/paper/panoptic-segformer","slug":"panoptic-segformer","title":"Panoptic SegFormer: Delving Deeper into Panoptic Segmentation with Transformers","date":"2021-09-08","arxiv_id":"2109.03814","n_code_links":3,"syntology":null},{"paper":null,"slug":"retrieve-caption-generate-visual-grounding","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","date":"2021-09-08","arxiv_id":"2109.03892","n_code_links":0,"syntology":null},{"paper":"/paper/what-s-hidden-in-a-one-layer-randomly","slug":"what-s-hidden-in-a-one-layer-randomly","title":"What's Hidden in a One-layer Randomly Weighted Transformer?","date":"2021-09-08","arxiv_id":"2109.03939","n_code_links":1,"syntology":null},{"paper":"/paper/fuseformer-fusing-fine-grained-information-in","slug":"fuseformer-fusing-fine-grained-information-in","title":"FuseFormer: Fusing Fine-Grained Information in Transformers for Video Inpainting","date":"2021-09-07","arxiv_id":"2109.02974","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":10,"n_instrument":0,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ruiliu-ai/fuseformer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":8,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gcst-graph-convolutional-skeleton-transformer","title":"Hierarchical Graph Convolutional Skeleton Transformer for Action Recognition","date":"2021-09-07","arxiv_id":"2109.02860","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixed-attention-transformer-for","title":"Mixed Attention Transformer for Leveraging Word-Level Knowledge to Neural Cross-Lingual Information Retrieval","date":"2021-09-07","arxiv_id":"2109.02789","n_code_links":0,"syntology":null},{"paper":"/paper/nnformer-interleaved-transformer-for","slug":"nnformer-interleaved-transformer-for","title":"nnFormer: Interleaved Transformer for Volumetric Segmentation","date":"2021-09-07","arxiv_id":"2109.03201","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["282857341/nnformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"puzzle-solving-without-search-or-human","title":"Puzzle Solving without Search or Human Knowledge: An Unnatural Language Approach","date":"2021-09-07","arxiv_id":"2109.02797","n_code_links":0,"syntology":null},{"paper":"/paper/rendezvous-attention-mechanisms-for-the","slug":"rendezvous-attention-mechanisms-for-the","title":"Rendezvous: Attention Mechanisms for the Recognition of Surgical Action Triplets in Endoscopic Videos","date":"2021-09-07","arxiv_id":"2109.03223","n_code_links":8,"syntology":null},{"paper":"/paper/3d-human-texture-estimation-from-a-single","slug":"3d-human-texture-estimation-from-a-single","title":"3D Human Texture Estimation from a Single Image with Transformers","date":"2021-09-06","arxiv_id":"2109.02563","n_code_links":1,"syntology":{"ran":14,"of":16,"n_ran_checked":7,"n_instrument":7,"unverified":2,"pointer_only":16,"phrase":"14 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","official":{"repos":["xuxy09/texformer"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/eliminating-sentiment-bias-for-aspect-level","slug":"eliminating-sentiment-bias-for-aspect-level","title":"Eliminating Sentiment Bias for Aspect-Level Sentiment Classification with Unsupervised Opinion Extraction","date":"2021-09-06","arxiv_id":"2109.02403","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-language-models-with-plug-and-play","slug":"enhancing-language-models-with-plug-and-play","title":"Enhancing Natural Language Representation with Large-Scale Out-of-Domain Commonsense","date":"2021-09-06","arxiv_id":"2109.02572","n_code_links":1,"syntology":null},{"paper":"/paper/permuteformer-efficient-relative-position","slug":"permuteformer-efficient-relative-position","title":"PermuteFormer: Efficient Relative Position Encoding for Long Sequences","date":"2021-09-06","arxiv_id":"2109.02377","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["cpcp1998/permuteformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-animation-transformer-visual","title":"The Animation Transformer: Visual Correspondence via Segment Matching","date":"2021-09-06","arxiv_id":"2109.02614","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformers-for-weeds-and-crops","title":"Vision Transformers For Weeds and Crops Classification Of High Resolution UAV Images","date":"2021-09-06","arxiv_id":"2109.02716","n_code_links":0,"syntology":null},{"paper":"/paper/voxel-transformer-for-3d-object-detection","slug":"voxel-transformer-for-3d-object-detection","title":"Voxel Transformer for 3D Object Detection","date":"2021-09-06","arxiv_id":"2109.02497","n_code_links":1,"syntology":{"ran":0,"of":7,"n_ran_checked":0,"n_instrument":0,"unverified":7,"pointer_only":7,"phrase":"0 ran · 7 unverified","official":null}},{"paper":"/paper/transformer-models-for-text-coherence","slug":"transformer-models-for-text-coherence","title":"Transformer Models for Text Coherence Assessment","date":"2021-09-05","arxiv_id":"2109.02176","n_code_links":2,"syntology":null},{"paper":null,"slug":"error-detection-in-large-scale-natural","title":"Error Detection in Large-Scale Natural Language Understanding Systems Using Transformer Models","date":"2021-09-04","arxiv_id":"2109.01754","n_code_links":0,"syntology":null},{"paper":null,"slug":"contextualized-embeddings-based-convolutional","title":"Contextualized Embeddings based Convolutional Neural Networks for Duplicate Question Identification","date":"2021-09-03","arxiv_id":"2109.01560","n_code_links":0,"syntology":null}],"record_sha256":"eb864d8e554cc3d12cf0b0f53f710c28efc9b530fa5f9d8911902676d603d32b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}