{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/attention/papers/252","list_of":"/method/attention","method":"Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":252,"pages_in_order":316,"rows_per_page":100,"rows":[25101,25200],"of":31583,"counts":{"archive_papers_tagged":31583,"with_a_code_link":13473,"where_syntology_ran_a_sample":3998,"not_listed_spam_title":0,"listed":31583,"listed_where_code_ran":3998,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3366,"every_run_a_failure_of_syntologys_instrument":632,"listed_with_a_run_with_no_instrument_failure":3366,"listed_every_run_a_failure_of_syntologys_instrument":632,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/attention","prev":"/method/attention/papers/251","next":"/method/attention/papers/253","papers":[{"paper":null,"slug":"an-empirical-study-on-transfer-learning-for","title":"An Empirical Study on Transfer Learning for Privilege Review","date":"2021-12-16","arxiv_id":"2112.08606","n_code_links":0,"syntology":null},{"paper":"/paper/block-skim-efficient-question-answering-for","slug":"block-skim-efficient-question-answering-for","title":"Block-Skim: Efficient Question Answering for Transformer","date":"2021-12-16","arxiv_id":"2112.08560","n_code_links":1,"syntology":null},{"paper":"/paper/call-for-customized-conversation-customized","slug":"call-for-customized-conversation-customized","title":"Call for Customized Conversation: Customized Conversation Grounding Persona and Knowledge","date":"2021-12-16","arxiv_id":"2112.08619","n_code_links":3,"syntology":null},{"paper":"/paper/commonsense-knowledge-augmented-pretrained-1","slug":"commonsense-knowledge-augmented-pretrained-1","title":"Knowledge-Augmented Language Models for Cause-Effect Relation Classification","date":"2021-12-16","arxiv_id":"2112.08615","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["phosseini/causal-reasoning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/crosssum-beyond-english-centric-cross-lingual","slug":"crosssum-beyond-english-centric-cross-lingual","title":"CrossSum: Beyond English-Centric Cross-Lingual Summarization for 1,500+ Language Pairs","date":"2021-12-16","arxiv_id":"2112.08804","n_code_links":1,"syntology":null},{"paper":null,"slug":"does-pre-training-induce-systematic-inference","title":"Does Pre-training Induce Systematic Inference? How Masked Language Models Acquire Commonsense Knowledge","date":"2021-12-16","arxiv_id":"2112.08583","n_code_links":0,"syntology":null},{"paper":"/paper/dprost-6-dof-object-pose-estimation-using","slug":"dprost-6-dof-object-pose-estimation-using","title":"DProST: Dynamic Projective Spatial Transformer Network for 6D Pose Estimation","date":"2021-12-16","arxiv_id":"2112.08775","n_code_links":1,"syntology":null},{"paper":"/paper/dream-uncovering-mental-models-behind","slug":"dream-uncovering-mental-models-behind","title":"DREAM: Improving Situational QA by First Elaborating the Situation","date":"2021-12-16","arxiv_id":"2112.08656","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/dream"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/efficient-hierarchical-domain-adaptation-for","slug":"efficient-hierarchical-domain-adaptation-for","title":"Efficient Hierarchical Domain Adaptation for Pretrained Language Models","date":"2021-12-16","arxiv_id":"2112.08786","n_code_links":1,"syntology":null},{"paper":null,"slug":"few-shot-semantic-parsing-with-language","title":"Few-Shot Semantic Parsing with Language Models Trained On Code","date":"2021-12-16","arxiv_id":"2112.08696","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-to-augment-your-vits-consistency-loss-and","title":"How to augment your ViTs? Consistency loss and StyleAug, a random style transfer augmentation","date":"2021-12-16","arxiv_id":"2112.09260","n_code_links":0,"syntology":null},{"paper":"/paper/kat-a-knowledge-augmented-transformer-for","slug":"kat-a-knowledge-augmented-transformer-for","title":"KAT: A Knowledge Augmented Transformer for Vision-and-Language","date":"2021-12-16","arxiv_id":"2112.08614","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-enhanced-session-based","title":"Knowledge-enhanced Session-based Recommendation with Temporal Transformer","date":"2021-12-16","arxiv_id":"2112.08745","n_code_links":0,"syntology":null},{"paper":"/paper/learning-bounded-context-free-grammar-via","slug":"learning-bounded-context-free-grammar-via","title":"Learning Bounded Context-Free-Grammar via LSTM and the Transformer:Difference and Explanations","date":"2021-12-16","arxiv_id":"2112.09174","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shihui2010/learn_cfg_with_neural_network"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-rich-representation-of-keyphrases-1","slug":"learning-rich-representation-of-keyphrases-1","title":"Learning Rich Representation of Keyphrases from Text","date":"2021-12-16","arxiv_id":"2112.08547","n_code_links":1,"syntology":null},{"paper":"/paper/looking-outside-the-box-to-ground-language-in","slug":"looking-outside-the-box-to-ground-language-in","title":"Bottom Up Top Down Detection Transformers for Language Grounding in Images and Point Clouds","date":"2021-12-16","arxiv_id":"2112.08879","n_code_links":1,"syntology":null},{"paper":null,"slug":"multivariate-realized-volatility-forecasting","title":"Multivariate Realized Volatility Forecasting with Graph Neural Network","date":"2021-12-16","arxiv_id":"2112.09015","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconsidering-the-past-optimizing-hidden-1","title":"Reconsidering the Past: Optimizing Hidden States in Language Models","date":"2021-12-16","arxiv_id":"2112.08653","n_code_links":0,"syntology":null},{"paper":"/paper/reframing-human-ai-collaboration-for","slug":"reframing-human-ai-collaboration-for","title":"Reframing Human-AI Collaboration for Generating Free-Text Explanations","date":"2021-12-16","arxiv_id":"2112.08674","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/few_shot_explanations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sgeitl-scene-graph-enhanced-image-text","title":"SGEITL: Scene Graph Enhanced Image-Text Learning for Visual Commonsense Reasoning","date":"2021-12-16","arxiv_id":"2112.08587","n_code_links":0,"syntology":null},{"paper":"/paper/trading-with-the-momentum-transformer-an","slug":"trading-with-the-momentum-transformer-an","title":"Trading with the Momentum Transformer: An Intelligent and Interpretable Architecture","date":"2021-12-16","arxiv_id":"2112.08534","n_code_links":3,"syntology":null},{"paper":"/paper/transzero-cross-attribute-guided-transformer","slug":"transzero-cross-attribute-guided-transformer","title":"TransZero++: Cross Attribute-Guided Transformer for Zero-Shot Learning","date":"2021-12-16","arxiv_id":"2112.08643","n_code_links":1,"syntology":null},{"paper":null,"slug":"trees-in-transformers-a-theoretical-analysis","title":"Trees in transformers: a theoretical analysis of the Transformer's ability to represent trees","date":"2021-12-16","arxiv_id":"2112.11913","n_code_links":0,"syntology":null},{"paper":null,"slug":"3d-question-answering","title":"3D Question Answering","date":"2021-12-15","arxiv_id":"2112.08359","n_code_links":0,"syntology":null},{"paper":null,"slug":"allwoz-towards-multilingual-task-oriented","title":"AllWOZ: Towards Multilingual Task-Oriented Dialog Systems for All","date":"2021-12-15","arxiv_id":"2112.08333","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-softtriple-loss-for-supervised","title":"Applying SoftTriple Loss for Supervised Language Model Fine Tuning","date":"2021-12-15","arxiv_id":"2112.08462","n_code_links":0,"syntology":null},{"paper":"/paper/dense-video-captioning-using-unsupervised","slug":"dense-video-captioning-using-unsupervised","title":"Dense Video Captioning Using Unsupervised Semantic Information","date":"2021-12-15","arxiv_id":"2112.08455","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-large-neural-language-models-for","title":"Fine-Tuning Large Neural Language Models for Biomedical Natural Language Processing","date":"2021-12-15","arxiv_id":"2112.07869","n_code_links":0,"syntology":null},{"paper":"/paper/is-my-favorite-new-movie-my-favorite-movie","slug":"is-my-favorite-new-movie-my-favorite-movie","title":"Is \"My Favorite New Movie\" My Favorite Movie? Probing the Understanding of Recursive Noun Phrases","date":"2021-12-15","arxiv_id":"2112.08326","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-transpile-amr-into-sparql","title":"Learning to Transpile AMR into SPARQL","date":"2021-12-15","arxiv_id":"2112.07877","n_code_links":0,"syntology":null},{"paper":null,"slug":"lesan-machine-translation-for-low-resource","title":"Lesan -- Machine Translation for Low Resource Languages","date":"2021-12-15","arxiv_id":"2112.08191","n_code_links":0,"syntology":null},{"paper":"/paper/longt5-efficient-text-to-text-transformer-for","slug":"longt5-efficient-text-to-text-transformer-for","title":"LongT5: Efficient Text-To-Text Transformer for Long Sequences","date":"2021-12-15","arxiv_id":"2112.07916","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/longt5"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/named-entity-recognition-architecture","slug":"named-entity-recognition-architecture","title":"Named entity recognition architecture combining contextual and global features","date":"2021-12-15","arxiv_id":"2112.08033","n_code_links":1,"syntology":null},{"paper":"/paper/one-size-does-not-fit-all-investigating","slug":"one-size-does-not-fit-all-investigating","title":"One size does not fit all: Investigating strategies for differentially-private learning across NLP tasks","date":"2021-12-15","arxiv_id":"2112.08159","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["trusthlt/dp-across-nlp-tasks"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"one-system-to-rule-them-all-a-universal","title":"One System to Rule them All: a Universal Intent Recognition System for Customer Service Chatbots","date":"2021-12-15","arxiv_id":"2112.08261","n_code_links":0,"syntology":null},{"paper":"/paper/oracle-linguistic-graphs-complement-a","slug":"oracle-linguistic-graphs-complement-a","title":"Linguistic Frameworks Go Toe-to-Toe at Neuro-Symbolic Language Modeling","date":"2021-12-15","arxiv_id":"2112.07874","n_code_links":1,"syntology":null},{"paper":"/paper/seqformer-a-frustratingly-simple-model-for","slug":"seqformer-a-frustratingly-simple-model-for","title":"SeqFormer: Sequential Transformer for Video Instance Segmentation","date":"2021-12-15","arxiv_id":"2112.08275","n_code_links":2,"syntology":null},{"paper":"/paper/spts-single-point-text-spotting","slug":"spts-single-point-text-spotting","title":"SPTS: Single-Point Text Spotting","date":"2021-12-15","arxiv_id":"2112.07917","n_code_links":1,"syntology":null},{"paper":null,"slug":"tracing-text-provenance-via-context-aware","title":"Tracing Text Provenance via Context-Aware Lexical Substitution","date":"2021-12-15","arxiv_id":"2112.07873","n_code_links":0,"syntology":null},{"paper":null,"slug":"vision-transformer-based-video-hashing","title":"Vision Transformer Based Video Hashing Retrieval for Tracing the Source of Fake Videos","date":"2021-12-15","arxiv_id":"2112.08117","n_code_links":0,"syntology":null},{"paper":null,"slug":"ace-bert-adversarial-cross-modal-enhanced","title":"ACE-BERT: Adversarial Cross-modal Enhanced BERT for E-commerce Retrieval","date":"2021-12-14","arxiv_id":"2112.07209","n_code_links":0,"syntology":null},{"paper":"/paper/adavit-adaptive-tokens-for-efficient-vision","slug":"adavit-adaptive-tokens-for-efficient-vision","title":"AdaViT: Adaptive Tokens for Efficient Vision Transformer","date":"2021-12-14","arxiv_id":"2112.07658","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"building-on-huang-et-al-glossbert-for-word","title":"Building on Huang et al. GlossBERT for Word Sense Disambiguation","date":"2021-12-14","arxiv_id":"2112.07089","n_code_links":0,"syntology":null},{"paper":null,"slug":"classifying-emails-into-human-vs-machine","title":"Classifying Emails into Human vs Machine Category","date":"2021-12-14","arxiv_id":"2112.07742","n_code_links":0,"syntology":null},{"paper":null,"slug":"coco-bert-improving-video-language-pre","title":"CoCo-BERT: Improving Video-Language Pre-training with Contrastive Cross-modal Matching and Denoising","date":"2021-12-14","arxiv_id":"2112.07515","n_code_links":0,"syntology":null},{"paper":null,"slug":"epigenomic-language-models-powered-by","title":"Epigenomic language models powered by Cerebras","date":"2021-12-14","arxiv_id":"2112.07571","n_code_links":0,"syntology":null},{"paper":"/paper/from-dense-to-sparse-contrastive-pruning-for","slug":"from-dense-to-sparse-contrastive-pruning-for","title":"From Dense to Sparse: Contrastive Pruning for Better Pre-trained Language Model Compression","date":"2021-12-14","arxiv_id":"2112.07198","n_code_links":2,"syntology":null},{"paper":"/paper/geometry-contrastive-transformer-for","slug":"geometry-contrastive-transformer-for","title":"Geometry-Contrastive Transformer for Generalized 3D Pose Transfer","date":"2021-12-14","arxiv_id":"2112.07374","n_code_links":1,"syntology":null},{"paper":"/paper/gpl-generative-pseudo-labeling-for","slug":"gpl-generative-pseudo-labeling-for","title":"GPL: Generative Pseudo Labeling for Unsupervised Domain Adaptation of Dense Retrieval","date":"2021-12-14","arxiv_id":"2112.07577","n_code_links":5,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/gpl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-compositional-generalization-with-2","slug":"improving-compositional-generalization-with-2","title":"Improving Compositional Generalization with Latent Structure and Data Augmentation","date":"2021-12-14","arxiv_id":"2112.07610","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-hybrid-ctc-attention-end-to-end","title":"Improving Hybrid CTC/Attention End-to-end Speech Recognition with Pretrained Acoustic and Language Model","date":"2021-12-14","arxiv_id":"2112.07254","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-fairness-with-biased-rulers-a","slug":"measuring-fairness-with-biased-rulers-a","title":"Measuring Fairness with Biased Rulers: A Survey on Quantifying Biases in Pretrained Language Models","date":"2021-12-14","arxiv_id":"2112.07447","n_code_links":1,"syntology":null},{"paper":null,"slug":"temporal-transformer-networks-with-self","title":"Temporal Transformer Networks with Self-Supervision for Action Recognition","date":"2021-12-14","arxiv_id":"2112.07338","n_code_links":0,"syntology":null},{"paper":"/paper/text-classification-models-for-form-entity","slug":"text-classification-models-for-form-entity","title":"Text Classification Models for Form Entity Linking","date":"2021-12-14","arxiv_id":"2112.07443","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-a-unified-foundation-model-jointly","title":"Towards a Unified Foundation Model: Jointly Pre-Training Transformers on Unpaired Images and Text","date":"2021-12-14","arxiv_id":"2112.07074","n_code_links":0,"syntology":null},{"paper":null,"slug":"5th-place-solution-for-vspw-2021-challenge","title":"5th Place Solution for VSPW 2021 Challenge","date":"2021-12-13","arxiv_id":"2112.06379","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-token-pruning-for-colbert","title":"A Study on Token Pruning for ColBERT","date":"2021-12-13","arxiv_id":"2112.06540","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-vs-target-word-quantifying-biases-in","title":"Measuring Context-Word Biases in Lexical Semantic Datasets","date":"2021-12-13","arxiv_id":"2112.06733","n_code_links":0,"syntology":null},{"paper":"/paper/dependency-learning-for-legal-judgment","slug":"dependency-learning-for-legal-judgment","title":"Dependency Learning for Legal Judgment Prediction with a Unified Text-to-Text Transformer","date":"2021-12-13","arxiv_id":"2112.06370","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-data-based-curricula-work","title":"Do Data-based Curricula Work?","date":"2021-12-13","arxiv_id":"2112.06510","n_code_links":0,"syntology":null},{"paper":"/paper/embracing-single-stride-3d-object-detector","slug":"embracing-single-stride-3d-object-detector","title":"Embracing Single Stride 3D Object Detector with Sparse Transformer","date":"2021-12-13","arxiv_id":"2112.06375","n_code_links":2,"syntology":null},{"paper":"/paper/glam-efficient-scaling-of-language-models","slug":"glam-efficient-scaling-of-language-models","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","date":"2021-12-13","arxiv_id":"2112.06905","n_code_links":0,"syntology":null},{"paper":null,"slug":"hformer-hybrid-cnn-transformer-for-fringe","title":"Hformer: Hybrid CNN-Transformer for Fringe Order Prediction in Phase Unwrapping of Fringe Projection","date":"2021-12-13","arxiv_id":"2112.06759","n_code_links":0,"syntology":null},{"paper":"/paper/keyphrase-generation-beyond-the-boundaries-of","slug":"keyphrase-generation-beyond-the-boundaries-of","title":"Keyphrase Generation Beyond the Boundaries of Title and Abstract","date":"2021-12-13","arxiv_id":"2112.06776","n_code_links":1,"syntology":null},{"paper":null,"slug":"pedestrian-trajectory-prediction-via-spatial","title":"Pedestrian Trajectory Prediction via Spatial Interaction Transformer Network","date":"2021-12-13","arxiv_id":"2112.06624","n_code_links":0,"syntology":null},{"paper":null,"slug":"roof-bert-divide-understanding-labour-and","title":"Roof-Transformer: Divided and Joined Understanding with Knowledge Enhancement","date":"2021-12-13","arxiv_id":"2112.06736","n_code_links":0,"syntology":null},{"paper":"/paper/sequential-recommendation-with-bidirectional","slug":"sequential-recommendation-with-bidirectional","title":"Improving Sequential Recommendations via Bidirectional Temporal Data Augmentation with Pre-training","date":"2021-12-13","arxiv_id":"2112.06460","n_code_links":1,"syntology":null},{"paper":"/paper/wechsel-effective-initialization-of-subword-1","slug":"wechsel-effective-initialization-of-subword-1","title":"WECHSEL: Effective initialization of subword embeddings for cross-lingual transfer of monolingual language models","date":"2021-12-13","arxiv_id":"2112.06598","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cpjku/wechsel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/implicit-transformer-network-for-screen-1","slug":"implicit-transformer-network-for-screen-1","title":"Implicit Transformer Network for Screen Content Image Continuous Super-Resolution","date":"2021-12-12","arxiv_id":"2112.06174","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["codyshen0000/itsrn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-logical-level-natural-language","title":"Improving Logical-Level Natural Language Generation with Topic-Conditioned Data Augmentation and Logical Form Generation","date":"2021-12-12","arxiv_id":"2112.06240","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-vision-transformers-for-incremental","title":"Improving Vision Transformers for Incremental Learning","date":"2021-12-12","arxiv_id":"2112.06103","n_code_links":0,"syntology":null},{"paper":"/paper/towards-more-efficient-insertion-transformer","slug":"towards-more-efficient-insertion-transformer","title":"Towards More Efficient Insertion Transformer with Fractional Positional Encoding","date":"2021-12-12","arxiv_id":"2112.06295","n_code_links":1,"syntology":null},{"paper":"/paper/composer-compositional-learning-of-group","slug":"composer-compositional-learning-of-group","title":"COMPOSER: Compositional Reasoning of Group Activity in Videos with Keypoint-Only Modality","date":"2021-12-11","arxiv_id":"2112.05892","n_code_links":1,"syntology":null},{"paper":null,"slug":"building-a-great-multi-lingual-teacher-with","title":"Building a great multi-lingual teacher with sparsely-gated mixture of experts for speech recognition","date":"2021-12-10","arxiv_id":"2112.05820","n_code_links":0,"syntology":null},{"paper":"/paper/couplformer-rethinking-vision-transformer","slug":"couplformer-rethinking-vision-transformer","title":"Couplformer:Rethinking Vision Transformer with Coupling Attention Map","date":"2021-12-10","arxiv_id":"2112.05425","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":2,"n_instrument":4,"unverified":6,"pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["wer010/Couplformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/deep-vit-features-as-dense-visual-descriptors","slug":"deep-vit-features-as-dense-visual-descriptors","title":"Deep ViT Features as Dense Visual Descriptors","date":"2021-12-10","arxiv_id":"2112.05814","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"findings-on-conversation-disentanglement","title":"Findings on Conversation Disentanglement","date":"2021-12-10","arxiv_id":"2112.05346","n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-interactions-using-pretrained","slug":"multimodal-interactions-using-pretrained","title":"Multimodal Interactions Using Pretrained Unimodal Models for SIMMC 2.0","date":"2021-12-10","arxiv_id":"2112.05328","n_code_links":1,"syntology":null},{"paper":"/paper/pre-training-and-fine-tuning-transformers-for","slug":"pre-training-and-fine-tuning-transformers-for","title":"Self-Supervised Transformers for fMRI representation","date":"2021-12-10","arxiv_id":"2112.05761","n_code_links":2,"syntology":null},{"paper":"/paper/self-attention-does-not-need-o-n-2-memory","slug":"self-attention-does-not-need-o-n-2-memory","title":"Self-attention Does Not Need $O(n^2)$ Memory","date":"2021-12-10","arxiv_id":"2112.05682","n_code_links":17,"syntology":{"ran":33,"of":36,"n_ran_checked":24,"n_instrument":9,"unverified":3,"pointer_only":8,"phrase":"33 ran (of which 0 constructed an object rather than computing a result; 24 with no instrument failure: 1 honoured, 6 violated, 17 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":{"repos":["google-research/google-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper","unlocated"]}}},{"paper":"/paper/sketching-as-a-tool-for-understanding-and","slug":"sketching-as-a-tool-for-understanding-and","title":"Sketching as a Tool for Understanding and Accelerating Self-attention for Long Sequences","date":"2021-12-10","arxiv_id":"2112.05359","n_code_links":1,"syntology":null},{"paper":null,"slug":"vut-versatile-ui-transformer-for-multi-modal","title":"VUT: Versatile UI Transformer for Multi-Modal Multi-Task User Interface Modeling","date":"2021-12-10","arxiv_id":"2112.05692","n_code_links":0,"syntology":null},{"paper":"/paper/3d-medical-point-transformer-introducing","slug":"3d-medical-point-transformer-introducing","title":"3D Medical Point Transformer: Introducing Convolution to Attention Networks for Medical Point Cloud Analysis","date":"2021-12-09","arxiv_id":"2112.04863","n_code_links":1,"syntology":null},{"paper":"/paper/a-bilingual-openworld-video-text-dataset-and","slug":"a-bilingual-openworld-video-text-dataset-and","title":"A Bilingual, OpenWorld Video Text Dataset and End-to-end Video Text Spotter with Transformer","date":"2021-12-09","arxiv_id":"2112.04888","n_code_links":3,"syntology":{"ran":16,"of":19,"n_ran_checked":14,"n_instrument":2,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 9 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["weijiawu/BOVText-Benchmark","weijiawu/transvtspotter"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-potentially-harmful-and-protective","slug":"detecting-potentially-harmful-and-protective","title":"Detecting potentially harmful and protective suicide-related content on twitter: A machine learning approach","date":"2021-12-09","arxiv_id":"2112.04796","n_code_links":2,"syntology":null},{"paper":null,"slug":"extending-adamw-by-leveraging-its-second","title":"Extending AdamW by Leveraging Its Second Moment and Magnitude","date":"2021-12-09","arxiv_id":"2112.06125","n_code_links":0,"syntology":null},{"paper":"/paper/fast-point-transformer","slug":"fast-point-transformer","title":"Fast Point Transformer","date":"2021-12-09","arxiv_id":"2112.04702","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["POSTECH-CVLab/FastPointTransformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"from-scattered-sources-to-comprehensive","title":"From Scattered Sources to Comprehensive Technology Landscape: A Recommendation-based Retrieval Approach","date":"2021-12-09","arxiv_id":"2112.04810","n_code_links":0,"syntology":null},{"paper":"/paper/injecting-semantic-concepts-into-end-to-end","slug":"injecting-semantic-concepts-into-end-to-end","title":"Injecting Semantic Concepts into End-to-End Image Captioning","date":"2021-12-09","arxiv_id":"2112.05230","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jacobswan1/ViTCAP"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pe-former-pose-estimation-transformer","slug":"pe-former-pose-estimation-transformer","title":"PE-former: Pose Estimation Transformer","date":"2021-12-09","arxiv_id":"2112.04981","n_code_links":1,"syntology":null},{"paper":"/paper/recurrent-glimpse-based-decoder-for-detection","slug":"recurrent-glimpse-based-decoder-for-detection","title":"Recurrent Glimpse-based Decoder for Detection with Transformer","date":"2021-12-09","arxiv_id":"2112.04632","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-search-as-extractive-paraphrase-span-1","slug":"semantic-search-as-extractive-paraphrase-span-1","title":"Semantic Search as Extractive Paraphrase Span Detection","date":"2021-12-09","arxiv_id":"2112.04886","n_code_links":1,"syntology":null},{"paper":"/paper/semi-supervised-medical-image-segmentation-2","slug":"semi-supervised-medical-image-segmentation-2","title":"Semi-Supervised Medical Image Segmentation via Cross Teaching between CNN and Transformer","date":"2021-12-09","arxiv_id":"2112.04894","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":1,"n_instrument":4,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HiLab-git/SSL4MIS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"towards-neural-functional-program-evaluation-1","title":"Towards Neural Functional Program Evaluation","date":"2021-12-09","arxiv_id":"2112.04630","n_code_links":0,"syntology":null},{"paper":"/paper/garment4d-garment-reconstruction-from-point-1","slug":"garment4d-garment-reconstruction-from-point-1","title":"Garment4D: Garment Reconstruction from Point Cloud Sequences","date":"2021-12-08","arxiv_id":"2112.04159","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hongfz16/garment4d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-language-models-by-retrieving-from","slug":"improving-language-models-by-retrieving-from","title":"Improving language models by retrieving from trillions of tokens","date":"2021-12-08","arxiv_id":"2112.04426","n_code_links":2,"syntology":{"ran":16,"of":23,"n_ran_checked":14,"n_instrument":2,"unverified":7,"pointer_only":3,"phrase":"16 ran (of which 5 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 3 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","official":null}},{"paper":"/paper/jaber-junior-arabic-bert","slug":"jaber-junior-arabic-bert","title":"JABER and SABER: Junior and Senior Arabic BERt","date":"2021-12-08","arxiv_id":"2112.04329","n_code_links":1,"syntology":null},{"paper":"/paper/joint-global-and-local-hierarchical-priors","slug":"joint-global-and-local-hierarchical-priors","title":"Joint Global and Local Hierarchical Priors for Learned Image Compression","date":"2021-12-08","arxiv_id":"2112.04487","n_code_links":1,"syntology":null},{"paper":"/paper/staf-a-spatio-temporal-attention-fusion","slug":"staf-a-spatio-temporal-attention-fusion","title":"MASTAF: A Model-Agnostic Spatio-Temporal Attention Fusion Network for Few-shot Video Classification","date":"2021-12-08","arxiv_id":"2112.04585","n_code_links":1,"syntology":null},{"paper":"/paper/transformaly-two-feature-spaces-are-better","slug":"transformaly-two-feature-spaces-are-better","title":"Transformaly -- Two (Feature Spaces) Are Better Than One","date":"2021-12-08","arxiv_id":"2112.04185","n_code_links":1,"syntology":null}],"record_sha256":"773955871b3c45f61dcbfa19994578f9b28491092dd3ead3a1a2f0da31921b07","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}