{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/transformer/papers/115","list_of":"/method/transformer","method":"Transformer","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":115,"pages_in_order":140,"rows_per_page":100,"rows":[11401,11500],"of":13999,"counts":{"archive_papers_tagged":13999,"with_a_code_link":6572,"where_syntology_ran_a_sample":2248,"not_listed_spam_title":0,"listed":13999,"listed_where_code_ran":2248,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1919,"every_run_a_failure_of_syntologys_instrument":329,"listed_with_a_run_with_no_instrument_failure":1919,"listed_every_run_a_failure_of_syntologys_instrument":329,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/transformer","prev":"/method/transformer/papers/114","next":"/method/transformer/papers/116","papers":[{"paper":"/paper/actions-speak-louder-than-listening","slug":"actions-speak-louder-than-listening","title":"Actions Speak Louder than Listening: Evaluating Music Style Transfer based on Editing Experience","date":"2021-10-25","arxiv_id":"2110.12855","n_code_links":1,"syntology":null},{"paper":"/paper/doctr-document-image-transformer-for","slug":"doctr-document-image-transformer-for","title":"DocTr: Document Image Transformer for Geometric Unwarping and Illumination Correction","date":"2021-10-25","arxiv_id":"2110.12942","n_code_links":2,"syntology":null},{"paper":null,"slug":"gophormer-ego-graph-transformer-for-node","title":"Gophormer: Ego-Graph Transformer for Node Classification","date":"2021-10-25","arxiv_id":"2110.13094","n_code_links":0,"syntology":null},{"paper":"/paper/history-aware-multimodal-transformer-for","slug":"history-aware-multimodal-transformer-for","title":"History Aware Multimodal Transformer for Vision-and-Language Navigation","date":"2021-10-25","arxiv_id":"2110.13309","n_code_links":1,"syntology":{"ran":7,"of":15,"n_ran_checked":7,"n_instrument":0,"unverified":8,"pointer_only":0,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mvt-multi-view-vision-transformer-for-3d","slug":"mvt-multi-view-vision-transformer-for-3d","title":"MVT: Multi-view Vision Transformer for 3D Object Recognition","date":"2021-10-25","arxiv_id":"2110.13083","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shanshuo/MVT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"revisiting-cnn-for-highly-inflected-bengali","title":"Paradigm Shift in Language Modeling: Revisiting CNN for Modeling Sanskrit Originated Bengali and Hindi Language","date":"2021-10-25","arxiv_id":"2110.13032","n_code_links":0,"syntology":null},{"paper":null,"slug":"stransgan-an-empirical-study-on-transformer-1","title":"The Nuts and Bolts of Adopting Transformer in GANs","date":"2021-10-25","arxiv_id":"2110.13107","n_code_links":0,"syntology":null},{"paper":"/paper/cvt-assd-convolutional-vision-transformer","slug":"cvt-assd-convolutional-vision-transformer","title":"CvT-ASSD: Convolutional vision-Transformer Based Attentive Single Shot MultiBox Detector","date":"2021-10-24","arxiv_id":"2110.12364","n_code_links":1,"syntology":null},{"paper":"/paper/3d-anas-v2-grafting-transformer-module-on","slug":"3d-anas-v2-grafting-transformer-module-on","title":"Grafting Transformer on Automatically Designed Convolutional Neural Network for Hyperspectral Image Classification","date":"2021-10-21","arxiv_id":"2110.11084","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-acceleration-with-dynamic-sparse","title":"Transformer Acceleration with Dynamic Sparse Attention","date":"2021-10-21","arxiv_id":"2110.11299","n_code_links":0,"syntology":null},{"paper":null,"slug":"vis-top-visual-transformer-overlay-processor","title":"Vis-TOP: Visual Transformer Overlay Processor","date":"2021-10-21","arxiv_id":"2110.10957","n_code_links":0,"syntology":null},{"paper":null,"slug":"after-unet-axial-fusion-transformer-unet-for","title":"AFTer-UNet: Axial Fusion Transformer UNet for Medical Image Segmentation","date":"2021-10-20","arxiv_id":"2110.10403","n_code_links":0,"syntology":null},{"paper":"/paper/aniformer-data-driven-3d-animation-with","slug":"aniformer-data-driven-3d-animation-with","title":"AniFormer: Data-driven 3D Animation with Transformer","date":"2021-10-20","arxiv_id":"2110.10533","n_code_links":1,"syntology":null},{"paper":null,"slug":"continual-learning-in-multilingual-nmt-via","title":"Continual Learning in Multilingual NMT via Language-Specific Embeddings","date":"2021-10-20","arxiv_id":"2110.10478","n_code_links":0,"syntology":null},{"paper":null,"slug":"esod-edge-based-task-scheduling-for-object","title":"ESOD:Edge-based Task Scheduling for Object Detection","date":"2021-10-20","arxiv_id":"2110.11342","n_code_links":0,"syntology":null},{"paper":"/paper/few-shot-temporal-action-localization-with","slug":"few-shot-temporal-action-localization-with","title":"Few-Shot Temporal Action Localization with Query Adaptive Transformer","date":"2021-10-20","arxiv_id":"2110.10552","n_code_links":1,"syntology":null},{"paper":null,"slug":"sea-graph-shell-attention-in-graph-neural","title":"SEA: Graph Shell Attention in Graph Neural Networks","date":"2021-10-20","arxiv_id":"2110.10674","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-accurate-and-reliable-iris","title":"Toward Accurate and Reliable Iris Segmentation Using Uncertainty Learning","date":"2021-10-20","arxiv_id":"2110.10334","n_code_links":0,"syntology":null},{"paper":null,"slug":"vldeformer-learning-visual-semantic","title":"VLDeformer: Vision-Language Decomposed Transformer for Fast Cross-Modal Retrieval","date":"2021-10-20","arxiv_id":"2110.11338","n_code_links":0,"syntology":null},{"paper":"/paper/a-picture-is-worth-a-thousand-words-a-unified","slug":"a-picture-is-worth-a-thousand-words-a-unified","title":"A Picture is Worth a Thousand Words: A Unified System for Diverse Captions and Rich Images Generation","date":"2021-10-19","arxiv_id":"2110.09756","n_code_links":1,"syntology":null},{"paper":null,"slug":"accelerating-framework-of-transformer-by","title":"Accelerating Framework of Transformer by Hardware Design and Model Compression Co-Optimization","date":"2021-10-19","arxiv_id":"2110.10030","n_code_links":0,"syntology":null},{"paper":null,"slug":"bilateral-vit-for-robust-fovea-localization","title":"Bilateral-ViT for Robust Fovea Localization","date":"2021-10-19","arxiv_id":"2110.09860","n_code_links":0,"syntology":null},{"paper":null,"slug":"detectornet-transformer-enhanced-spatial","title":"DetectorNet: Transformer-enhanced Spatial Temporal Graph Neural Network for Traffic Prediction","date":"2021-10-19","arxiv_id":"2111.00869","n_code_links":0,"syntology":null},{"paper":"/paper/generating-symbolic-reasoning-problems-with-1","slug":"generating-symbolic-reasoning-problems-with-1","title":"Generating Symbolic Reasoning Problems with Transformer GANs","date":"2021-10-19","arxiv_id":"2110.10054","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":6,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["reactive-systems/TGAN-SR"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"inductive-biases-and-variable-creation-in-1","title":"Inductive Biases and Variable Creation in Self-Attention Mechanisms","date":"2021-10-19","arxiv_id":"2110.10090","n_code_links":0,"syntology":null},{"paper":"/paper/permutation-invariant-graph-to-sequence-model-1","slug":"permutation-invariant-graph-to-sequence-model-1","title":"Permutation invariant graph-to-sequence model for template-free retrosynthesis and reaction prediction","date":"2021-10-19","arxiv_id":"2110.09681","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["coleygroup/graph2smiles"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"spatial-temporal-transformer-for-3d-point","title":"Spatial-Temporal Transformer for 3D Point Cloud Sequences","date":"2021-10-19","arxiv_id":"2110.09783","n_code_links":0,"syntology":null},{"paper":"/paper/ssast-self-supervised-audio-spectrogram","slug":"ssast-self-supervised-audio-spectrogram","title":"SSAST: Self-Supervised Audio Spectrogram Transformer","date":"2021-10-19","arxiv_id":"2110.09784","n_code_links":3,"syntology":{"ran":13,"of":16,"n_ran_checked":13,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["YuanGongND/ssast"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/unifying-multimodal-transformer-for-bi","slug":"unifying-multimodal-transformer-for-bi","title":"Unifying Multimodal Transformer for Bi-directional Image and Text Generation","date":"2021-10-19","arxiv_id":"2110.09753","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["researchmm/generate-it"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/compositional-attention-disentangling-search-1","slug":"compositional-attention-disentangling-search-1","title":"Compositional Attention: Disentangling Search and Retrieval","date":"2021-10-18","arxiv_id":"2110.09419","n_code_links":3,"syntology":null},{"paper":"/paper/hrformer-high-resolution-transformer-for","slug":"hrformer-high-resolution-transformer-for","title":"HRFormer: High-Resolution Transformer for Dense Prediction","date":"2021-10-18","arxiv_id":"2110.09408","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 7 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 7 samples that ran constructed an object rather than computing a result","official":{"repos":["HRNet/HRFormer"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/sentimentarcs-a-novel-method-for-self","slug":"sentimentarcs-a-novel-method-for-self","title":"SentimentArcs: A Novel Method for Self-Supervised Sentiment Analysis of Time Series Shows SOTA Transformers Can Struggle Finding Narrative Arcs","date":"2021-10-18","arxiv_id":"2110.09454","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["jon-chun/sentimentarcs_notebooks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/sequential-modeling-with-multiple-attributes","slug":"sequential-modeling-with-multiple-attributes","title":"Sequential Modeling with Multiple Attributes for Watchlist Recommendation in E-Commerce","date":"2021-10-18","arxiv_id":"2110.11072","n_code_links":1,"syntology":null},{"paper":"/paper/3d-retr-end-to-end-single-and-multi-view-3d","slug":"3d-retr-end-to-end-single-and-multi-view-3d","title":"3D-RETR: End-to-End Single and Multi-View 3D Reconstruction with Transformers","date":"2021-10-17","arxiv_id":"2110.08861","n_code_links":1,"syntology":null},{"paper":null,"slug":"cae-transformer-transformer-based-model-to","title":"CAE-Transformer: Transformer-based Model to Predict Invasiveness of Lung Adenocarcinoma Subsolid Nodules from Non-thin Section 3D CT Scans","date":"2021-10-17","arxiv_id":"2110.08721","n_code_links":0,"syntology":null},{"paper":"/paper/siamese-transformer-pyramid-networks-for-real","slug":"siamese-transformer-pyramid-networks-for-real","title":"Siamese Transformer Pyramid Networks for Real-Time UAV Tracking","date":"2021-10-17","arxiv_id":"2110.08822","n_code_links":1,"syntology":null},{"paper":"/paper/a-good-prompt-is-worth-millions-of-parameters","slug":"a-good-prompt-is-worth-millions-of-parameters","title":"A Good Prompt Is Worth Millions of Parameters: Low-resource Prompt-based Learning for Vision-Language Models","date":"2021-10-16","arxiv_id":"2110.08484","n_code_links":1,"syntology":null},{"paper":null,"slug":"alleviating-the-inequality-of-attention-heads-1","title":"Alleviating the Inequality of Attention Heads for Neural Machine Translation","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/asformer-transformer-for-action-segmentation","slug":"asformer-transformer-for-action-segmentation","title":"ASFormer: Transformer for Action Segmentation","date":"2021-10-16","arxiv_id":"2110.08568","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chinayi/asformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attention-temperature-matters-in-abstractive-1","title":"Attention Temperature Matters in Abstractive Summarization Distillation","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"covid-19-detection-in-chest-x-ray-images-1","title":"COVID-19 Detection in Chest X-ray Images Using Swin-Transformer and Transformer in Transformer","date":"2021-10-16","arxiv_id":"2110.08427","n_code_links":0,"syntology":null},{"paper":null,"slug":"emotion-flip-reasoning-in-multiparty","title":"Emotion Flip Reasoning in Multiparty Conversations","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-transformer-networks-for-long","title":"Hierarchical Transformer Networks for Long-sequence and Multiple Clinical Documents Classification","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ode-transformer-an-ordinary-differential-1","title":"ODE Transformer: An Ordinary Differential Equation-Inspired Model for Sequence Generation","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/transformer-with-a-mixture-of-gaussian-keys-1","slug":"transformer-with-a-mixture-of-gaussian-keys-1","title":"Improving Transformers with Probabilistic Attention Keys","date":"2021-10-16","arxiv_id":"2110.08678","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minhtannguyen/transformer-mgk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"combining-cnns-with-transformer-for","title":"Combining CNNs With Transformer for Multimodal 3D MRI Brain Tumor Segmentation With Self-Supervised Pretraining","date":"2021-10-15","arxiv_id":"2110.07919","n_code_links":0,"syntology":null},{"paper":"/paper/on-learning-the-transformer-kernel-1","slug":"on-learning-the-transformer-kernel-1","title":"On Learning the Transformer Kernel","date":"2021-10-15","arxiv_id":"2110.08323","n_code_links":1,"syntology":null},{"paper":null,"slug":"streamult-streaming-multimodal-transformer","title":"StreaMulT: Streaming Multimodal Transformer for Heterogeneous and Arbitrary Long Sequential Data","date":"2021-10-15","arxiv_id":"2110.08021","n_code_links":0,"syntology":null},{"paper":null,"slug":"causal-transformers-perform-below-chance-on","title":"Causal Transformers Perform Below Chance on Recursive Nested Constructions, Unlike Humans","date":"2021-10-14","arxiv_id":"2110.07240","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-off-the-shelf-machine-listening","title":"Evaluating Off-the-Shelf Machine Listening and Natural Language Models for Automated Audio Captioning","date":"2021-10-14","arxiv_id":"2110.07410","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolutionary-trajectory-and-origin-of-sars","title":"Integrating Fréchet distance and AI reveals the evolutionary trajectory and origin of SARS-CoV-2","date":"2021-10-14","arxiv_id":"2110.07696","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-introductions-in-podcast-episodes","slug":"identifying-introductions-in-podcast-episodes","title":"Identifying Introductions in Podcast Episodes from Automatically Generated Transcripts","date":"2021-10-14","arxiv_id":"2110.07096","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-drug-target-interaction-prediction","title":"Improved Drug-target Interaction Prediction with Intermolecular Graph Transformer","date":"2021-10-14","arxiv_id":"2110.07347","n_code_links":0,"syntology":null},{"paper":"/paper/non-autoregressive-translation-with-layer","slug":"non-autoregressive-translation-with-layer","title":"Non-Autoregressive Translation with Layer-Wise Prediction and Deep Supervision","date":"2021-10-14","arxiv_id":"2110.07515","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenyangh/dslp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sub-word-level-lip-reading-with-visual","slug":"sub-word-level-lip-reading-with-visual","title":"Sub-word Level Lip Reading With Visual Attention","date":"2021-10-14","arxiv_id":"2110.07603","n_code_links":0,"syntology":null},{"paper":"/paper/the-neural-data-router-adaptive-control-flow","slug":"the-neural-data-router-adaptive-control-flow","title":"The Neural Data Router: Adaptive Control Flow in Transformers Improves Systematic Generalization","date":"2021-10-14","arxiv_id":"2110.07732","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["robertcsordas/ndr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"transformer-for-polyp-detection","title":"Transformer for Polyp Detection","date":"2021-10-14","arxiv_id":"2111.07918","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip4caption-clip-for-video-caption","title":"CLIP4Caption: CLIP for Video Caption","date":"2021-10-13","arxiv_id":"2110.06615","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-redundancy-in-attention-with-reuse-1","slug":"leveraging-redundancy-in-attention-with-reuse-1","title":"Leveraging redundancy in attention with Reuse Transformers","date":"2021-10-13","arxiv_id":"2110.06821","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantics-aware-attention-improves-neural","title":"Semantics-aware Attention Improves Neural Machine Translation","date":"2021-10-13","arxiv_id":"2110.06920","n_code_links":0,"syntology":null},{"paper":"/paper/study-of-positional-encoding-approaches-for","slug":"study-of-positional-encoding-approaches-for","title":"Study of positional encoding approaches for Audio Spectrogram Transformers","date":"2021-10-13","arxiv_id":"2110.06999","n_code_links":1,"syntology":null},{"paper":"/paper/the-dawn-of-quantum-natural-language","slug":"the-dawn-of-quantum-natural-language","title":"The Dawn of Quantum Natural Language Processing","date":"2021-10-13","arxiv_id":"2110.06510","n_code_links":2,"syntology":null},{"paper":null,"slug":"transform-and-bitstream-domain-image","title":"Transform and Bitstream Domain Image Classification","date":"2021-10-13","arxiv_id":"2110.06740","n_code_links":0,"syntology":null},{"paper":"/paper/yformer-u-net-inspired-transformer-1","slug":"yformer-u-net-inspired-transformer-1","title":"Yformer: U-Net Inspired Transformer Architecture for Far Horizon Time Series Forecasting","date":"2021-10-13","arxiv_id":"2110.08255","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-guided-generative-models-for","title":"Attention-guided Generative Models for Extractive Question Answering","date":"2021-10-12","arxiv_id":"2110.06393","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-learning-for-representation","slug":"contrastive-learning-for-representation","title":"Contrastive Learning for Representation Degeneration Problem in Sequential Recommendation","date":"2021-10-12","arxiv_id":"2110.05730","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["RuihongQiu/DuoRec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/discodvt-generating-long-text-with-discourse","slug":"discodvt-generating-long-text-with-discourse","title":"DiscoDVT: Generating Long Text with Discourse-Aware Discrete Variational Transformer","date":"2021-10-12","arxiv_id":"2110.05999","n_code_links":1,"syntology":null},{"paper":"/paper/hetformer-heterogeneous-transformer-with","slug":"hetformer-heterogeneous-transformer-with","title":"HETFORMER: Heterogeneous Transformer with Sparse Attention for Long-Text Extractive Summarization","date":"2021-10-12","arxiv_id":"2110.06388","n_code_links":1,"syntology":null},{"paper":"/paper/lightseq-accelerated-training-for-transformer","slug":"lightseq-accelerated-training-for-transformer","title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","date":"2021-10-12","arxiv_id":"2110.05722","n_code_links":1,"syntology":null},{"paper":"/paper/mention-memory-incorporating-textual-1","slug":"mention-memory-incorporating-textual-1","title":"Mention Memory: incorporating textual knowledge into Transformers through entity mention attention","date":"2021-10-12","arxiv_id":"2110.06176","n_code_links":1,"syntology":null},{"paper":"/paper/relative-molecule-self-attention-transformer-1","slug":"relative-molecule-self-attention-transformer-1","title":"Relative Molecule Self-Attention Transformer","date":"2021-10-12","arxiv_id":"2110.05841","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/rescoring-sequence-to-sequence-models-for","slug":"rescoring-sequence-to-sequence-models-for","title":"Rescoring Sequence-to-Sequence Models for Text Line Recognition with CTC-Prefixes","date":"2021-10-12","arxiv_id":"2110.05909","n_code_links":1,"syntology":null},{"paper":"/paper/satellite-image-semantic-segmentation","slug":"satellite-image-semantic-segmentation","title":"Satellite Image Semantic Segmentation","date":"2021-10-12","arxiv_id":"2110.05812","n_code_links":1,"syntology":null},{"paper":"/paper/starformer-transformer-with-state-action-1","slug":"starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.06206","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["elicassion/StARformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/adaptively-multi-view-and-temporal-fusing","slug":"adaptively-multi-view-and-temporal-fusing","title":"Adaptive Multi-view and Temporal Fusing Transformer for 3D Human Pose Estimation","date":"2021-10-11","arxiv_id":"2110.05092","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-transfer-learning-capabilities","slug":"investigating-transfer-learning-capabilities","title":"Investigating Transfer Learning Capabilities of Vision Transformers and CNNs by Fine-Tuning a Single Trainable Block","date":"2021-10-11","arxiv_id":"2110.05270","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-view-self-attention-based-transformer","title":"Multi-View Self-Attention Based Transformer for Speaker Recognition","date":"2021-10-11","arxiv_id":"2110.05036","n_code_links":0,"syntology":null},{"paper":null,"slug":"sru-pioneering-fast-recurrence-with-attention","title":"SRU++: Pioneering Fast Recurrence with Attention for Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05571","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-source-separation-via-bayesian","slug":"unsupervised-source-separation-via-bayesian","title":"Unsupervised Source Separation via Bayesian Inference in the Latent Domain","date":"2021-10-11","arxiv_id":"2110.05313","n_code_links":1,"syntology":null},{"paper":null,"slug":"dct-dynamic-compressive-transformer-for","title":"DCT: Dynamic Compressive Transformer for Modeling Unbounded Sequence","date":"2021-10-10","arxiv_id":"2110.04821","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-channel-end-to-end-neural-diarization","title":"Multi-Channel End-to-End Neural Diarization with Distributed Microphones","date":"2021-10-10","arxiv_id":"2110.04694","n_code_links":0,"syntology":null},{"paper":"/paper/nvit-vision-transformer-compression-and-1","slug":"nvit-vision-transformer-compression-and-1","title":"Global Vision Transformer Pruning with Hessian-Aware Saliency","date":"2021-10-10","arxiv_id":"2110.04869","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"on-automatic-text-extractive-summarization","title":"Automatic Text Extractive Summarization Based on Graph and Pre-trained Language Model Attention","date":"2021-10-10","arxiv_id":"2110.04878","n_code_links":0,"syntology":null},{"paper":"/paper/sp-gpt2-semantics-improvement-in-vietnamese","slug":"sp-gpt2-semantics-improvement-in-vietnamese","title":"SP-GPT2: Semantics Improvement in Vietnamese Poetry Generation","date":"2021-10-10","arxiv_id":"2110.15723","n_code_links":2,"syntology":null},{"paper":null,"slug":"supershaper-task-agnostic-super-pre-training","title":"SuperShaper: Task-Agnostic Super Pre-training of BERT Models with Variable Hidden Dimensions","date":"2021-10-10","arxiv_id":"2110.04711","n_code_links":0,"syntology":null},{"paper":"/paper/vector-quantized-image-modeling-with-improved-1","slug":"vector-quantized-image-modeling-with-improved-1","title":"Vector-quantized Image Modeling with Improved VQGAN","date":"2021-10-09","arxiv_id":"2110.04627","n_code_links":5,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"2110-06794","title":"The Layout Generation Algorithm of Graphic Design Based on Transformer-CVAE","date":"2021-10-08","arxiv_id":"2110.06794","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-token-attacks-on-vision","title":"Adversarial Token Attacks on Vision Transformers","date":"2021-10-08","arxiv_id":"2110.04337","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-lgm-leveraging-object-context","title":"Context-LGM: Leveraging Object-Context Relation for Context-Aware Object Recognition","date":"2021-10-08","arxiv_id":"2110.04042","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-learning-for-source-code-with","title":"Towards Learning (Dis)-Similarity of Source Code from Program Contrasts","date":"2021-10-08","arxiv_id":"2110.03868","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-enhanced-hierarchical-graph","slug":"knowledge-enhanced-hierarchical-graph","title":"Knowledge-Enhanced Hierarchical Graph Transformer Network for Multi-Behavior Recommendation","date":"2021-10-08","arxiv_id":"2110.04000","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-adaptive-control-flow-in","title":"Learning Adaptive Control Flow in Transformers for Improved Systematic Generalization","date":"2021-10-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"m6-10t-a-sharing-delinking-paradigm-for","title":"M6-10T: A Sharing-Delinking Paradigm for Efficient Multi-Trillion Parameter Pretraining","date":"2021-10-08","arxiv_id":"2110.03888","n_code_links":0,"syntology":null},{"paper":"/paper/multiplex-behavioral-relation-learning-for","slug":"multiplex-behavioral-relation-learning-for","title":"Multiplex Behavioral Relation Learning for Recommendation via Memory Augmented Transformer Network","date":"2021-10-08","arxiv_id":"2110.04002","n_code_links":1,"syntology":null},{"paper":"/paper/rpt-toward-transferable-model-on","slug":"rpt-toward-transferable-model-on","title":"RPT: Toward Transferable Model on Heterogeneous Researcher Data via Pre-Training","date":"2021-10-08","arxiv_id":"2110.07336","n_code_links":1,"syntology":null},{"paper":"/paper/taming-sparsely-activated-transformer-with","slug":"taming-sparsely-activated-transformer-with","title":"Taming Sparsely Activated Transformer with Stochastic Experts","date":"2021-10-08","arxiv_id":"2110.04260","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/stochastic-mixture-of-experts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vidt-an-efficient-and-effective-fully","slug":"vidt-an-efficient-and-effective-fully","title":"ViDT: An Efficient and Effective Fully Transformer-based Object Detector","date":"2021-10-08","arxiv_id":"2110.03921","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-is-all-you-need-good-embeddings","title":"Attention is All You Need? Good Embeddings with Statistics are enough:Large Scale Audio Understanding without Transformers/ Convolutions/ BERTs/ Mixers/ Attention/ RNNs or ....","date":"2021-10-07","arxiv_id":"2110.03183","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-supermask-pruning-learning-to","slug":"end-to-end-supermask-pruning-learning-to","title":"End-to-End Supermask Pruning: Learning to Prune Image Captioning Models","date":"2021-10-07","arxiv_id":"2110.03298","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiahuei/sparse-image-captioning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"f734a17fbbea5b5a808b68c2ea8e1af289975a8ad8e3b7493617eb49ad97bb40","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}