{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/178","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":178,"pages_in_order":249,"rows_per_page":100,"rows":[17701,17800],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/177","next":"/method/multi-head-attention/papers/179","papers":[{"paper":null,"slug":"simulation-driven-training-of-vision","title":"Simulation-Driven Training of Vision Transformers Enabling Metal Segmentation in X-Ray Images","date":"2022-03-17","arxiv_id":"2203.09207","n_code_links":0,"syntology":null},{"paper":null,"slug":"transframer-arbitrary-frame-prediction-with","title":"Transframer: Arbitrary Frame Prediction with Generative Models","date":"2022-03-17","arxiv_id":"2203.09494","n_code_links":0,"syntology":null},{"paper":"/paper/unimo-2-end-to-end-unified-vision-language","slug":"unimo-2-end-to-end-unified-vision-language","title":"UNIMO-2: End-to-End Unified Vision-Language Grounded Learning","date":"2022-03-17","arxiv_id":"2203.09067","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-squeeze-and-excitation-and-transformer","title":"A Squeeze-and-Excitation and Transformer based Cross-task System for Environmental Sound Recognition","date":"2022-03-16","arxiv_id":"2203.08350","n_code_links":0,"syntology":null},{"paper":"/paper/adapler-speeding-up-inference-by-adaptive-1","slug":"adapler-speeding-up-inference-by-adaptive-1","title":"AdapLeR: Speeding up Inference by Adaptive Length Reduction","date":"2022-03-16","arxiv_id":"2203.08991","n_code_links":1,"syntology":null},{"paper":"/paper/deciwatch-a-simple-baseline-for-10x-efficient","slug":"deciwatch-a-simple-baseline-for-10x-efficient","title":"DeciWatch: A Simple Baseline for 10x Efficient 2D and 3D Pose Estimation","date":"2022-03-16","arxiv_id":"2203.08713","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cure-lab/DeciWatch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/edter-edge-detection-with-transformer","slug":"edter-edge-detection-with-transformer","title":"EDTER: Edge Detection with Transformer","date":"2022-03-16","arxiv_id":"2203.08566","n_code_links":1,"syntology":null},{"paper":"/paper/kinyabert-a-morphology-aware-kinyarwanda-1","slug":"kinyabert-a-morphology-aware-kinyarwanda-1","title":"KinyaBERT: a Morphology-aware Kinyarwanda Language Model","date":"2022-03-16","arxiv_id":"2203.08459","n_code_links":1,"syntology":{"ran":6,"of":12,"n_ran_checked":2,"n_instrument":4,"unverified":6,"pointer_only":0,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["anzeyimana/kinyabert-acl2022"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/label-semantics-for-few-shot-named-entity","slug":"label-semantics-for-few-shot-named-entity","title":"Label Semantics for Few Shot Named Entity Recognition","date":"2022-03-16","arxiv_id":"2203.08985","n_code_links":1,"syntology":null},{"paper":"/paper/open-set-recognition-using-vision-transformer","slug":"open-set-recognition-using-vision-transformer","title":"Open Set Recognition using Vision Transformer with an Additional Detection Head","date":"2022-03-16","arxiv_id":"2203.08441","n_code_links":1,"syntology":null},{"paper":"/paper/the-devil-is-in-the-details-window-based","slug":"the-devil-is-in-the-details-window-based","title":"The Devil Is in the Details: Window-based Attention for Image Compression","date":"2022-03-16","arxiv_id":"2203.08450","n_code_links":2,"syntology":{"ran":15,"of":15,"n_ran_checked":11,"n_instrument":4,"unverified":0,"pointer_only":11,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["googolxx/stf"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/thinking-about-gpt-3-in-context-learning-for","slug":"thinking-about-gpt-3-in-context-learning-for","title":"Thinking about GPT-3 In-Context Learning for Biomedical IE? Think Again","date":"2022-03-16","arxiv_id":"2203.08410","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-practical-certifiable-patch-defense","title":"Towards Practical Certifiable Patch Defense with Vision Transformer","date":"2022-03-16","arxiv_id":"2203.08519","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-semantic-segmentation-by-2","slug":"unsupervised-semantic-segmentation-by-2","title":"Unsupervised Semantic Segmentation by Distilling Feature Correspondences","date":"2022-03-16","arxiv_id":"2203.08414","n_code_links":3,"syntology":{"ran":11,"of":14,"n_ran_checked":7,"n_instrument":4,"unverified":3,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mhamilton723/STEGO"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"wegformer-transformers-for-weakly-supervised","title":"WegFormer: Transformers for Weakly Supervised Semantic Segmentation","date":"2022-03-16","arxiv_id":"2203.08421","n_code_links":0,"syntology":null},{"paper":null,"slug":"2-speed-network-ensemble-for-efficient","title":"2-speed network ensemble for efficient classification of incremental land-use/land-cover satellite image chips","date":"2022-03-15","arxiv_id":"2203.08267","n_code_links":0,"syntology":null},{"paper":null,"slug":"actformer-a-gan-transformer-framework-towards","title":"ActFormer: A GAN-based Transformer towards General Action-Conditioned 3D Human Motion Generation","date":"2022-03-15","arxiv_id":"2203.07706","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-sentence-representation-for","slug":"compressing-sentence-representation-for","title":"Compressing Sentence Representation for Semantic Retrieval via Homomorphic Projective Distillation","date":"2022-03-15","arxiv_id":"2203.07687","n_code_links":1,"syntology":null},{"paper":"/paper/data-contamination-from-memorization-to-1","slug":"data-contamination-from-memorization-to-1","title":"Data Contamination: From Memorization to Exploitation","date":"2022-03-15","arxiv_id":"2203.08242","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["schwartz-lab-nlp/data_contamination"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/do-berts-learn-to-use-browser-user-interface-1","slug":"do-berts-learn-to-use-browser-user-interface-1","title":"Do BERTs Learn to Use Browser User Interface? Exploring Multi-Step Tasks with Unified Vision-and-Language BERTs","date":"2022-03-15","arxiv_id":"2203.07828","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["Alab-NII/bertbui_pub"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/do-language-models-plagiarize","slug":"do-language-models-plagiarize","title":"Do Language Models Plagiarize?","date":"2022-03-15","arxiv_id":"2203.07618","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-long-sequence-encoding-via-1","title":"Efficient Long Sequence Encoding via Synchronization","date":"2022-03-15","arxiv_id":"2203.07644","n_code_links":0,"syntology":null},{"paper":"/paper/fast-autofocusing-using-tiny-networks-for","slug":"fast-autofocusing-using-tiny-networks-for","title":"Fast Autofocusing using Tiny Transformer Networks for Digital Holographic Microscopy","date":"2022-03-15","arxiv_id":"2203.07772","n_code_links":1,"syntology":null},{"paper":null,"slug":"generating-privacy-preserving-process-data","title":"Generating Privacy-Preserving Process Data with Deep Generative Models","date":"2022-03-15","arxiv_id":"2203.07949","n_code_links":0,"syntology":null},{"paper":"/paper/humus-net-hybrid-unrolled-multi-scale-network","slug":"humus-net-hybrid-unrolled-multi-scale-network","title":"HUMUS-Net: Hybrid unrolled multi-scale network architecture for accelerated MRI reconstruction","date":"2022-03-15","arxiv_id":"2203.08213","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["MathFLDS/HUMUS-Net","z-fabian/HUMUS-Net"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/imputing-out-of-vocabulary-embeddings-with","slug":"imputing-out-of-vocabulary-embeddings-with","title":"Imputing Out-of-Vocabulary Embeddings with LOVE Makes Language Models Robust with Little Cost","date":"2022-03-15","arxiv_id":"2203.07860","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tigerchen52/love"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/inverted-pyramid-multi-task-transformer-for","slug":"inverted-pyramid-multi-task-transformer-for","title":"InvPT: Inverted Pyramid Multi-task Transformer for Dense Scene Understanding","date":"2022-03-15","arxiv_id":"2203.07997","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["prismformore/InvPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/smoothing-matters-momentum-transformer-for","slug":"smoothing-matters-momentum-transformer-for","title":"Smoothing Matters: Momentum Transformer for Domain Adaptive Semantic Segmentation","date":"2022-03-15","arxiv_id":"2203.07988","n_code_links":1,"syntology":{"ran":3,"of":8,"n_ran_checked":1,"n_instrument":2,"unverified":5,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["alpc91/transda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"the-ghost-in-the-machine-has-an-american","title":"The Ghost in the Machine has an American accent: value conflict in GPT-3","date":"2022-03-15","arxiv_id":"2203.07785","n_code_links":0,"syntology":null},{"paper":"/paper/unified-visual-transformer-compression-1","slug":"unified-visual-transformer-compression-1","title":"Unified Visual Transformer Compression","date":"2022-03-15","arxiv_id":"2203.08243","n_code_links":1,"syntology":{"ran":9,"of":16,"n_ran_checked":6,"n_instrument":3,"unverified":7,"pointer_only":3,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","official":{"repos":["VITA-Group/UVC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/accelerating-detr-convergence-via-semantic","slug":"accelerating-detr-convergence-via-semantic","title":"Accelerating DETR Convergence via Semantic-Aligned Matching","date":"2022-03-14","arxiv_id":"2203.06883","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zhanggongjie/sam-detr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/all-in-one-exploring-unified-video-language","slug":"all-in-one-exploring-unified-video-language","title":"All in One: Exploring Unified Video-Language Pre-training","date":"2022-03-14","arxiv_id":"2203.07303","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-pre-trained-transformers-be-used-in","title":"Can pre-trained Transformers be used in detecting complex sensitive sentences? -- A Monsanto case study","date":"2022-03-14","arxiv_id":"2203.06793","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-visual-semantic-pretraining","title":"Contrastive Visual Semantic Pretraining Magnifies the Semantics of Natural Language Representations","date":"2022-03-14","arxiv_id":"2203.07511","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-language-modeling-with-sparse-all","slug":"efficient-language-modeling-with-sparse-all","title":"Efficient Language Modeling with Sparse all-MLP","date":"2022-03-14","arxiv_id":"2203.06850","n_code_links":0,"syntology":null},{"paper":"/paper/eit-efficiently-lead-inductive-biases-to-vit","slug":"eit-efficiently-lead-inductive-biases-to-vit","title":"Deep Transformers Thirst for Comprehensive-Frequency Data","date":"2022-03-14","arxiv_id":"2203.07116","n_code_links":1,"syntology":null},{"paper":"/paper/grips-gradient-free-edit-based-instruction","slug":"grips-gradient-free-edit-based-instruction","title":"GrIPS: Gradient-free, Edit-based Instruction Search for Prompting Large Language Models","date":"2022-03-14","arxiv_id":"2203.07281","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["archiki/grips"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"switch-trajectory-transformer-with","title":"Switch Trajectory Transformer with Distributional Value Approximation for Multi-Task Reinforcement Learning","date":"2022-03-14","arxiv_id":"2203.07413","n_code_links":0,"syntology":null},{"paper":"/paper/the-optimal-bert-surgeon-scalable-and","slug":"the-optimal-bert-surgeon-scalable-and","title":"The Optimal BERT Surgeon: Scalable and Accurate Second-Order Pruning for Large Language Models","date":"2022-03-14","arxiv_id":"2203.07259","n_code_links":1,"syntology":null},{"paper":"/paper/vast-the-valence-assessing-semantics-test-for","slug":"vast-the-valence-assessing-semantics-test-for","title":"VAST: The Valence-Assessing Semantics Test for Contextualizing Language Models","date":"2022-03-14","arxiv_id":"2203.07504","n_code_links":1,"syntology":null},{"paper":null,"slug":"wcl-bbcd-a-contrastive-learning-and-knowledge","title":"WCL-BBCD: A Contrastive Learning and Knowledge Graph Approach to Named Entity Recognition","date":"2022-03-14","arxiv_id":"2203.06925","n_code_links":0,"syntology":null},{"paper":"/paper/cmkd-cnn-transformer-based-cross-model","slug":"cmkd-cnn-transformer-based-cross-model","title":"CMKD: CNN/Transformer-Based Cross-Model Knowledge Distillation for Audio Classification","date":"2022-03-13","arxiv_id":"2203.06760","n_code_links":2,"syntology":null},{"paper":null,"slug":"investigating-the-impact-of-covid-19-on","title":"Investigating the Impact of COVID-19 on Education by Social Network Mining","date":"2022-03-13","arxiv_id":"2203.06584","n_code_links":0,"syntology":null},{"paper":"/paper/masked-autoencoders-for-point-cloud-self","slug":"masked-autoencoders-for-point-cloud-self","title":"Masked Autoencoders for Point Cloud Self-supervised Learning","date":"2022-03-13","arxiv_id":"2203.06604","n_code_links":4,"syntology":{"ran":9,"of":9,"n_ran_checked":6,"n_instrument":3,"unverified":0,"pointer_only":7,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Pang-Yatian/Point-MAE"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"satr-slice-attention-with-transformer-for","title":"SATr: Slice Attention with Transformer for Universal Lesion Detection","date":"2022-03-13","arxiv_id":"2203.07373","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-up-your-kernels-to-31x31-revisiting","slug":"scaling-up-your-kernels-to-31x31-revisiting","title":"Scaling Up Your Kernels to 31x31: Revisiting Large Kernel Design in CNNs","date":"2022-03-13","arxiv_id":"2203.06717","n_code_links":8,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["DingXiaoH/RepLKNet-pytorch","megvii-research/replknet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/scinli-a-corpus-for-natural-language","slug":"scinli-a-corpus-for-natural-language","title":"SciNLI: A Corpus for Natural Language Inference on Scientific Text","date":"2022-03-13","arxiv_id":"2203.06728","n_code_links":1,"syntology":null},{"paper":"/paper/sparse-local-patch-transformer-for-robust","slug":"sparse-local-patch-transformer-for-robust","title":"Sparse Local Patch Transformer for Robust Face Alignment and Landmarks Inherent Relation Learning","date":"2022-03-13","arxiv_id":"2203.06541","n_code_links":1,"syntology":null},{"paper":"/paper/bibert-accurate-fully-binarized-bert-1","slug":"bibert-accurate-fully-binarized-bert-1","title":"BiBERT: Accurate Fully Binarized BERT","date":"2022-03-12","arxiv_id":"2203.06390","n_code_links":1,"syntology":null},{"paper":null,"slug":"dftr-depth-supervised-hierarchical-feature","title":"DFTR: Depth-supervised Fusion Transformer for Salient Object Detection","date":"2022-03-12","arxiv_id":"2203.06429","n_code_links":0,"syntology":null},{"paper":"/paper/elle-efficient-lifelong-pre-training-for-1","slug":"elle-efficient-lifelong-pre-training-for-1","title":"ELLE: Efficient Lifelong Pre-training for Emerging Data","date":"2022-03-12","arxiv_id":"2203.06311","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thunlp/elle"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/finer-financial-numeric-entity-recognition","slug":"finer-financial-numeric-entity-recognition","title":"FiNER: Financial Numeric Entity Recognition for XBRL Tagging","date":"2022-03-12","arxiv_id":"2203.06482","n_code_links":2,"syntology":null},{"paper":null,"slug":"joint-cnn-and-transformer-network-via-weakly","title":"Joint CNN and Transformer Network via weakly supervised Learning for efficient crowd counting","date":"2022-03-12","arxiv_id":"2203.06388","n_code_links":0,"syntology":null},{"paper":"/paper/markbert-marking-word-boundaries-improves-1","slug":"markbert-marking-word-boundaries-improves-1","title":"MarkBERT: Marking Word Boundaries Improves Chinese BERT","date":"2022-03-12","arxiv_id":"2203.06378","n_code_links":1,"syntology":null},{"paper":null,"slug":"one-stage-video-instance-segmentation-from","title":"One-stage Video Instance Segmentation: From Frame-in Frame-out to Clip-in Clip-out","date":"2022-03-12","arxiv_id":"2203.06421","n_code_links":0,"syntology":null},{"paper":"/paper/the-principle-of-diversity-training-stronger","slug":"the-principle-of-diversity-training-stronger","title":"The Principle of Diversity: Training Stronger Vision Transformers Calls for Reducing All Levels of Redundancy","date":"2022-03-12","arxiv_id":"2203.06345","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["vita-group/diverse-vit"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/wasserstein-adversarial-transformer-for-cloud","slug":"wasserstein-adversarial-transformer-for-cloud","title":"Wasserstein Adversarial Transformer for Cloud Workload Prediction","date":"2022-03-12","arxiv_id":"2203.06501","n_code_links":1,"syntology":null},{"paper":"/paper/a-sentence-is-worth-128-pseudo-tokens-a-1","slug":"a-sentence-is-worth-128-pseudo-tokens-a-1","title":"A Sentence is Worth 128 Pseudo Tokens: A Semantic-Aware Contrastive Learning Framework for Sentence Embeddings","date":"2022-03-11","arxiv_id":"2203.05877","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["namco0816/pt-bert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/bertopic-neural-topic-modeling-with-a-class","slug":"bertopic-neural-topic-modeling-with-a-class","title":"BERTopic: Neural topic modeling with a class-based TF-IDF procedure","date":"2022-03-11","arxiv_id":"2203.05794","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":4,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["MaartenGr/BERTopic","maartengr/bertopic_evaluation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/block-recurrent-transformers","slug":"block-recurrent-transformers","title":"Block-Recurrent Transformers","date":"2022-03-11","arxiv_id":"2203.07852","n_code_links":3,"syntology":{"ran":18,"of":23,"n_ran_checked":9,"n_instrument":9,"unverified":5,"pointer_only":4,"phrase":"18 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 3 violated, 4 with no contract checked; 9 where Syntology's instrument failed) · 5 unverified","official":{"repos":["google-research/meliad"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/block-sparse-adversarial-attack-to-fool","slug":"block-sparse-adversarial-attack-to-fool","title":"Block-Sparse Adversarial Attack to Fool Transformer-Based Text Classifiers","date":"2022-03-11","arxiv_id":"2203.05948","n_code_links":1,"syntology":null},{"paper":null,"slug":"font-shape-to-impression-translation","title":"Font Shape-to-Impression Translation","date":"2022-03-11","arxiv_id":"2203.05808","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-bert-for-medical-document","title":"Hierarchical BERT for Medical Document Understanding","date":"2022-03-11","arxiv_id":"2204.09600","n_code_links":0,"syntology":null},{"paper":null,"slug":"pathsage-spatial-graph-attention-neural","title":"PathSAGE: Spatial Graph Attention Neural Networks With Random Path Sampling","date":"2022-03-11","arxiv_id":"2203.05793","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-based-streaming-asr-with","title":"Transformer-based Streaming ASR with Cumulative Attention","date":"2022-03-11","arxiv_id":"2203.05736","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-word-embeddings-to-analyze-protests","title":"Using Word Embeddings to Analyze Protests News","date":"2022-03-11","arxiv_id":"2203.05875","n_code_links":0,"syntology":null},{"paper":"/paper/verbert-automating-brazilian-case-law","slug":"verbert-automating-brazilian-case-law","title":"verBERT: Automating Brazilian Case Law Document Multi-label Categorization Using BERT","date":"2022-03-11","arxiv_id":"2203.06224","n_code_links":1,"syntology":null},{"paper":null,"slug":"visualizing-and-understanding-patch","title":"Visualizing and Understanding Patch Interactions in Vision Transformer","date":"2022-03-11","arxiv_id":"2203.05922","n_code_links":0,"syntology":null},{"paper":"/paper/when-classifying-grammatical-role-bert-doesn-1","slug":"when-classifying-grammatical-role-bert-doesn-1","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2022-03-11","arxiv_id":"2203.06204","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-new-approach-to-calculating-bertscore-for","title":"A new approach to calculating BERTScore for automatic assessment of translation quality","date":"2022-03-10","arxiv_id":"2203.05598","n_code_links":0,"syntology":null},{"paper":"/paper/contextualized-sensorimotor-norms-multi-1","slug":"contextualized-sensorimotor-norms-multi-1","title":"Contextualized Sensorimotor Norms: multi-dimensional measures of sensorimotor strength for ambiguous English words, in context","date":"2022-03-10","arxiv_id":"2203.05648","n_code_links":1,"syntology":null},{"paper":null,"slug":"look-backward-and-forward-self-knowledge","title":"Look Backward and Forward: Self-Knowledge Distillation with Bidirectional Decoder for Neural Machine Translation","date":"2022-03-10","arxiv_id":"2203.05248","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-free-attentive-scoring-for-speaker","slug":"parameter-free-attentive-scoring-for-speaker","title":"Parameter-Free Attentive Scoring for Speaker Verification","date":"2022-03-10","arxiv_id":"2203.05642","n_code_links":1,"syntology":null},{"paper":null,"slug":"semantic-norm-recognition-and-its-application","title":"Semantic Norm Recognition and its application to Portuguese Law","date":"2022-03-10","arxiv_id":"2203.05425","n_code_links":0,"syntology":null},{"paper":"/paper/speciesist-language-and-nonhuman-animal-bias-1","slug":"speciesist-language-and-nonhuman-animal-bias-1","title":"Speciesist Language and Nonhuman Animal Bias in English Masked Language Models","date":"2022-03-10","arxiv_id":"2203.05140","n_code_links":1,"syntology":null},{"paper":null,"slug":"stylebabel-artistic-style-tagging-and","title":"StyleBabel: Artistic Style Tagging and Captioning","date":"2022-03-10","arxiv_id":"2203.05321","n_code_links":0,"syntology":null},{"paper":"/paper/truetype-transformer-character-and-font-style","slug":"truetype-transformer-character-and-font-style","title":"TrueType Transformer: Character and Font Style Recognition in Outline Format","date":"2022-03-10","arxiv_id":"2203.05338","n_code_links":1,"syntology":null},{"paper":"/paper/anti-oversmoothing-in-deep-vision","slug":"anti-oversmoothing-in-deep-vision","title":"Anti-Oversmoothing in Deep Vision Transformers via the Fourier Domain Analysis: From Theory to Practice","date":"2022-03-09","arxiv_id":"2203.05962","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["vita-group/vit-anti-oversmoothing"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/chitransformer-towards-reliable-stereo-from","slug":"chitransformer-towards-reliable-stereo-from","title":"ChiTransformer:Towards Reliable Stereo from Cues","date":"2022-03-09","arxiv_id":"2203.04554","n_code_links":1,"syntology":null},{"paper":"/paper/coarse-to-fine-sparse-transformer-for","slug":"coarse-to-fine-sparse-transformer-for","title":"Coarse-to-Fine Sparse Transformer for Hyperspectral Image Reconstruction","date":"2022-03-09","arxiv_id":"2203.04845","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":9,"n_instrument":4,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":{"repos":["caiyuanhao1998/MST"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":8,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"detecting-and-diagnosing-terrestrial","title":"Detecting and Diagnosing Terrestrial Gravitational-Wave Mimics Through Feature Learning","date":"2022-03-09","arxiv_id":"2203.05086","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiscale-convolutional-transformer-with","title":"Multiscale Convolutional Transformer with Center Mask Pretraining for Hyperspectral Image Classification","date":"2022-03-09","arxiv_id":"2203.04771","n_code_links":0,"syntology":null},{"paper":"/paper/nlx-gpt-a-model-for-natural-language","slug":"nlx-gpt-a-model-for-natural-language","title":"NLX-GPT: A Model for Natural Language Explanations in Vision and Vision-Language Tasks","date":"2022-03-09","arxiv_id":"2203.05081","n_code_links":1,"syntology":null},{"paper":"/paper/phtrans-parallelly-aggregating-global-and","slug":"phtrans-parallelly-aggregating-global-and","title":"PHTrans: Parallelly Aggregating Global and Local Representations for Medical Image Segmentation","date":"2022-03-09","arxiv_id":"2203.04568","n_code_links":2,"syntology":null},{"paper":null,"slug":"region-aware-face-swapping","title":"Region-Aware Face Swapping","date":"2022-03-09","arxiv_id":"2203.04564","n_code_links":0,"syntology":null},{"paper":"/paper/the-evolution-evolvability-and-engineering-of","slug":"the-evolution-evolvability-and-engineering-of","title":"The evolution, evolvability and engineering of gene regulatory DNA","date":"2022-03-09","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"uni4eye-unified-2d-and-3d-self-supervised-pre","title":"Uni4Eye: Unified 2D and 3D Self-supervised Pre-training via Masked Image Modeling Transformer for Ophthalmic Image Classification","date":"2022-03-09","arxiv_id":"2203.04614","n_code_links":0,"syntology":null},{"paper":null,"slug":"cass-a-channel-aware-self-supervised","title":"CaSS: A Channel-aware Self-supervised Representation Learning Framework for Multivariate Time Series Classification","date":"2022-03-08","arxiv_id":"2203.04298","n_code_links":0,"syntology":null},{"paper":"/paper/coarse-to-fine-vision-transformer","slug":"coarse-to-fine-vision-transformer","title":"CF-ViT: A General Coarse-to-Fine Method for Vision Transformer","date":"2022-03-08","arxiv_id":"2203.03821","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["chenmnz/cf-vit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dumlp-pin-a-dual-mlp-dot-product-permutation","slug":"dumlp-pin-a-dual-mlp-dot-product-permutation","title":"DuMLP-Pin: A Dual-MLP-dot-product Permutation-invariant Network for Set Feature Extraction","date":"2022-03-08","arxiv_id":"2203.04007","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-group-transformer-a-general-vision","title":"Dynamic Group Transformer: A General Vision Transformer Backbone with Dynamic Group Attention","date":"2022-03-08","arxiv_id":"2203.03937","n_code_links":0,"syntology":null},{"paper":"/paper/edgeformer-improving-light-weight-convnets-by","slug":"edgeformer-improving-light-weight-convnets-by","title":"ParC-Net: Position Aware Circular Convolution with Merits from ConvNets and Transformer","date":"2022-03-08","arxiv_id":"2203.03952","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hkzhang91/edgeformer","hkzhang91/pacc-net","hkzhang91/parc-net"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/graph-attention-transformer-network-for-multi","slug":"graph-attention-transformer-network-for-multi","title":"Graph Attention Transformer Network for Multi-Label Image Classification","date":"2022-03-08","arxiv_id":"2203.04049","n_code_links":1,"syntology":null},{"paper":"/paper/joint-rotational-invariance-and-adversarial","slug":"joint-rotational-invariance-and-adversarial","title":"Joint rotational invariance and adversarial training of a dual-stream Transformer yields state of the art Brain-Score for Area V4","date":"2022-03-08","arxiv_id":"2203.06649","n_code_links":1,"syntology":null},{"paper":"/paper/lane-detection-with-versatile-atrousformer","slug":"lane-detection-with-versatile-atrousformer","title":"Lane Detection with Versatile AtrousFormer and Local Semantic Guidance","date":"2022-03-08","arxiv_id":"2203.04067","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-the-mixing-of-contextual","slug":"measuring-the-mixing-of-contextual","title":"Measuring the Mixing of Contextual Information in the Transformer","date":"2022-03-08","arxiv_id":"2203.04212","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mt-upc/transformer-contributions"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/neural-face-identification-in-a-2d-wireframe-1","slug":"neural-face-identification-in-a-2d-wireframe-1","title":"Neural Face Identification in a 2D Wireframe Projection of a Manifold Object","date":"2022-03-08","arxiv_id":"2203.04229","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":5,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["manycore-research/faceformer"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-generalized-models-for-task-oriented","title":"Towards Generalized Models for Task-oriented Dialogue Modeling on Spoken Conversations","date":"2022-03-08","arxiv_id":"2203.04045","n_code_links":0,"syntology":null},{"paper":"/paper/it5-large-scale-text-to-text-pretraining-for","slug":"it5-large-scale-text-to-text-pretraining-for","title":"IT5: Text-to-text Pretraining for Italian Language Understanding and Generation","date":"2022-03-07","arxiv_id":"2203.03759","n_code_links":3,"syntology":null},{"paper":null,"slug":"monocular-robot-navigation-with-self","title":"Monocular Robot Navigation with Self-Supervised Pretrained Vision Transformers","date":"2022-03-07","arxiv_id":"2203.03682","n_code_links":0,"syntology":null}],"record_sha256":"8895b743c375e0965fa8f36e1d173c3fdec7eb829571cbcef8c84149d88f6157","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}