{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/multi-head-attention/papers/159","list_of":"/method/multi-head-attention","method":"Multi-Head Attention","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":159,"pages_in_order":249,"rows_per_page":100,"rows":[15801,15900],"of":24855,"counts":{"archive_papers_tagged":24855,"with_a_code_link":11214,"where_syntology_ran_a_sample":3454,"not_listed_spam_title":0,"listed":24855,"listed_where_code_ran":3454,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2916,"every_run_a_failure_of_syntologys_instrument":538,"listed_with_a_run_with_no_instrument_failure":2916,"listed_every_run_a_failure_of_syntologys_instrument":538,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/multi-head-attention","prev":"/method/multi-head-attention/papers/158","next":"/method/multi-head-attention/papers/160","papers":[{"paper":"/paper/convtransseg-a-multi-resolution-convolution","slug":"convtransseg-a-multi-resolution-convolution","title":"ConvTransSeg: A Multi-resolution Convolution-Transformer Network for Medical Image Segmentation","date":"2022-10-13","arxiv_id":"2210.07072","n_code_links":1,"syntology":null},{"paper":"/paper/demystifying-self-supervised-trojan-attacks","slug":"demystifying-self-supervised-trojan-attacks","title":"An Embarrassingly Simple Backdoor Attack on Self-supervised Learning","date":"2022-10-13","arxiv_id":"2210.07346","n_code_links":4,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["meet-cjli/ctrl","CCCjiang/CTRL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"explanations-from-large-language-models-make","title":"Explanations from Large Language Models Make Small Reasoners Better","date":"2022-10-13","arxiv_id":"2210.06726","n_code_links":0,"syntology":null},{"paper":"/paper/feature-proxy-transformer-for-few-shot","slug":"feature-proxy-transformer-for-few-shot","title":"Feature-Proxy Transformer for Few-Shot Segmentation","date":"2022-10-13","arxiv_id":"2210.06908","n_code_links":2,"syntology":null},{"paper":"/paper/how-to-train-vision-transformer-on-small","slug":"how-to-train-vision-transformer-on-small","title":"How to Train Vision Transformer on Small-scale Datasets?","date":"2022-10-13","arxiv_id":"2210.07240","n_code_links":2,"syntology":null},{"paper":"/paper/intermediate-prototype-mining-transformer-for","slug":"intermediate-prototype-mining-transformer-for","title":"Intermediate Prototype Mining Transformer for Few-Shot Semantic Segmentation","date":"2022-10-13","arxiv_id":"2210.06780","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":6,"n_instrument":3,"unverified":4,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["liuyuanwei98/ipmt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/joint-reasoning-on-hybrid-knowledge-sources","slug":"joint-reasoning-on-hybrid-knowledge-sources","title":"Joint Reasoning on Hybrid-knowledge sources for Task-Oriented Dialog","date":"2022-10-13","arxiv_id":"2210.07295","n_code_links":1,"syntology":null},{"paper":null,"slug":"jointly-reinforced-user-simulator-and-task-1","title":"Jointly Reinforced User Simulator and Task-oriented Dialog System with Simplified Generative Architecture","date":"2022-10-13","arxiv_id":"2210.06706","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-of-code-are-few-shot","slug":"language-models-of-code-are-few-shot","title":"Language Models of Code are Few-Shot Commonsense Learners","date":"2022-10-13","arxiv_id":"2210.07128","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["madaan/cocogen"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/large-language-models-are-few-1-shot-table","slug":"large-language-models-are-few-1-shot-table","title":"Large Language Models are few(1)-shot Table Reasoners","date":"2022-10-13","arxiv_id":"2210.06710","n_code_links":1,"syntology":null},{"paper":"/paper/overlooked-video-classification-in-weakly","slug":"overlooked-video-classification-in-weakly","title":"Overlooked Video Classification in Weakly Supervised Video Anomaly Detection","date":"2022-10-13","arxiv_id":"2210.06688","n_code_links":1,"syntology":null},{"paper":"/paper/q-vit-accurate-and-fully-quantized-low-bit","slug":"q-vit-accurate-and-fully-quantized-low-bit","title":"Q-ViT: Accurate and Fully Quantized Low-bit Vision Transformer","date":"2022-10-13","arxiv_id":"2210.06707","n_code_links":1,"syntology":null},{"paper":null,"slug":"scene-text-image-super-resolution-via-content","title":"Scene Text Image Super-Resolution via Content Perceptual Loss and Criss-Cross Transformer Blocks","date":"2022-10-13","arxiv_id":"2210.06924","n_code_links":0,"syntology":null},{"paper":null,"slug":"squat-sharpness-and-quantization-aware","title":"SQuAT: Sharpness- and Quantization-Aware Training for BERT","date":"2022-10-13","arxiv_id":"2210.07171","n_code_links":0,"syntology":null},{"paper":null,"slug":"swformer-sparse-window-transformer-for-3d","title":"SWFormer: Sparse Window Transformer for 3D Object Detection in Point Clouds","date":"2022-10-13","arxiv_id":"2210.07372","n_code_links":0,"syntology":null},{"paper":null,"slug":"tone-prediction-and-orthographic-conversion","title":"Tone prediction and orthographic conversion for Basaa","date":"2022-10-13","arxiv_id":"2210.06986","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-sample-efficient-nlp-models-more-robust","title":"Are Sample-Efficient NLP Models More Robust?","date":"2022-10-12","arxiv_id":"2210.06456","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-vision-transformers","slug":"bridging-the-gap-between-vision-transformers","title":"Bridging the Gap Between Vision Transformers and Convolutional Neural Networks on Small Datasets","date":"2022-10-12","arxiv_id":"2210.05958","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["arieseirack/dhvt"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ctl-evaluating-generalization-on-never-seen","slug":"ctl-evaluating-generalization-on-never-seen","title":"CTL++: Evaluating Generalization on Never-Seen Compositional Patterns of Known Functions, and Compatibility of Neural Representations","date":"2022-10-12","arxiv_id":"2210.06350","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["robertcsordas/ctlpp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"datscore-evaluating-translation-with-data","title":"DATScore: Evaluating Translation with Data Augmented Translations","date":"2022-10-12","arxiv_id":"2210.06576","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-clustering-network-for-unsupervised","title":"ACSeg: Adaptive Conceptualization for Unsupervised Semantic Segmentation","date":"2022-10-12","arxiv_id":"2210.05944","n_code_links":0,"syntology":null},{"paper":"/paper/flare7k-a-phenomenological-nighttime-flare","slug":"flare7k-a-phenomenological-nighttime-flare","title":"Flare7K: A Phenomenological Nighttime Flare Removal Dataset","date":"2022-10-12","arxiv_id":"2210.06570","n_code_links":1,"syntology":null},{"paper":null,"slug":"fonttransformer-few-shot-high-resolution","title":"FontTransformer: Few-shot High-resolution Chinese Glyph Image Synthesis via Stacked Transformers","date":"2022-10-12","arxiv_id":"2210.06301","n_code_links":0,"syntology":null},{"paper":"/paper/foundation-transformers","slug":"foundation-transformers","title":"Foundation Transformers","date":"2022-10-12","arxiv_id":"2210.06423","n_code_links":4,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/frustratingly-simple-entity-tracking-with","slug":"frustratingly-simple-entity-tracking-with","title":"Entity Tracking via Effective Use of Multi-Task Learning Model and Mention-guided Decoding","date":"2022-10-12","arxiv_id":"2210.06444","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iamjanvijay/meet","iamjanvijay/set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gmp-well-tuned-global-magnitude-pruning-can","title":"GMP*: Well-Tuned Gradual Magnitude Pruning Can Outperform Most BERT-Pruning Methods","date":"2022-10-12","arxiv_id":"2210.06384","n_code_links":0,"syntology":null},{"paper":"/paper/hate-clipper-multimodal-hateful-meme","slug":"hate-clipper-multimodal-hateful-meme","title":"Hate-CLIPper: Multimodal Hateful Meme Classification based on Cross-modal Interaction of CLIP Features","date":"2022-10-12","arxiv_id":"2210.05916","n_code_links":1,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"0 ran · 5 unverified","official":{"repos":["gokulkarthik/hateclipper"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":null,"slug":"improved-data-augmentation-for-translation","title":"Improved Data Augmentation for Translation Suggestion","date":"2022-10-12","arxiv_id":"2210.06138","n_code_links":0,"syntology":null},{"paper":"/paper/instruction-tuning-for-few-shot-aspect-based","slug":"instruction-tuning-for-few-shot-aspect-based","title":"Instruction Tuning for Few-Shot Aspect-Based Sentiment Analysis","date":"2022-10-12","arxiv_id":"2210.06629","n_code_links":1,"syntology":null},{"paper":"/paper/jukedrummer-conditional-beat-aware-audio","slug":"jukedrummer-conditional-beat-aware-audio","title":"JukeDrummer: Conditional Beat-aware Audio-domain Drum Accompaniment Generation via Transformer VQ-VAE","date":"2022-10-12","arxiv_id":"2210.06007","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-models-are-parsimonious-learners","title":"The Lazy Neuron Phenomenon: On Emergence of Activation Sparsity in Transformers","date":"2022-10-12","arxiv_id":"2210.06313","n_code_links":0,"syntology":null},{"paper":"/paper/long-form-video-language-pre-training-with","slug":"long-form-video-language-pre-training-with","title":"Long-Form Video-Language Pre-Training with Multimodal Temporal Contrastive Learning","date":"2022-10-12","arxiv_id":"2210.06031","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/xpretrain"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/motionbert-unified-pretraining-for-human","slug":"motionbert-unified-pretraining-for-human","title":"MotionBERT: A Unified Perspective on Learning Human Motion Representations","date":"2022-10-12","arxiv_id":"2210.06551","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":6,"n_instrument":2,"unverified":3,"pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Walter0807/MotionBERT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/non-autoregressive-machine-translation-with-2","slug":"non-autoregressive-machine-translation-with-2","title":"Integrating Translation Memories into Non-Autoregressive Machine Translation","date":"2022-10-12","arxiv_id":"2210.06020","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-text-style-transfer-via-style-masked","title":"On Text Style Transfer via Style Masked Language Models","date":"2022-10-12","arxiv_id":"2210.06394","n_code_links":0,"syntology":null},{"paper":"/paper/predictive-querying-for-autoregressive-neural","slug":"predictive-querying-for-autoregressive-neural","title":"Predictive Querying for Autoregressive Neural Sequence Models","date":"2022-10-12","arxiv_id":"2210.06464","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ajboyd2/prob_seq_queries"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/prepended-domain-transformer-heterogeneous","slug":"prepended-domain-transformer-heterogeneous","title":"Prepended Domain Transformer: Heterogeneous Face Recognition without Bells and Whistles","date":"2022-10-12","arxiv_id":"2210.06529","n_code_links":2,"syntology":null},{"paper":"/paper/probing-commonsense-knowledge-in-pre-trained","slug":"probing-commonsense-knowledge-in-pre-trained","title":"Probing Commonsense Knowledge in Pre-trained Language Models with Sense-level Precision and Expanded Vocabulary","date":"2022-10-12","arxiv_id":"2210.06376","n_code_links":1,"syntology":null},{"paper":null,"slug":"rankt5-fine-tuning-t5-for-text-ranking-with","title":"RankT5: Fine-Tuning T5 for Text Ranking with Ranking Losses","date":"2022-10-12","arxiv_id":"2210.10634","n_code_links":0,"syntology":null},{"paper":"/paper/s4nd-modeling-images-and-videos-as","slug":"s4nd-modeling-images-and-videos-as","title":"S4ND: Modeling Images and Videos as Multidimensional Signals Using State Spaces","date":"2022-10-12","arxiv_id":"2210.06583","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hazyresearch/state-spaces"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":null,"slug":"sumbot-summarizing-context-in-open-domain","title":"SUMBot: Summarizing Context in Open-Domain Dialogue Systems","date":"2022-10-12","arxiv_id":"2210.06496","n_code_links":0,"syntology":null},{"paper":"/paper/towards-theoretically-inspired-neural","slug":"towards-theoretically-inspired-neural","title":"Towards Theoretically Inspired Neural Initialization Optimization","date":"2022-10-12","arxiv_id":"2210.05956","n_code_links":1,"syntology":null},{"paper":"/paper/uplift-and-upsample-efficient-3d-human-pose","slug":"uplift-and-upsample-efficient-3d-human-pose","title":"Uplift and Upsample: Efficient 3D Human Pose Estimation with Uplifting Transformers","date":"2022-10-12","arxiv_id":"2210.06110","n_code_links":2,"syntology":{"ran":15,"of":17,"n_ran_checked":15,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["goldbricklemon/uplift-upsample-3dhpe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/zits-image-inpainting-by-improving-the","slug":"zits-image-inpainting-by-improving-the","title":"ZITS++: Image Inpainting by Improving the Incremental Transformer on Structural Priors","date":"2022-10-12","arxiv_id":"2210.05950","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ewrfcas/zits-plusplus","dqiaole/zits_inpainting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-win-win-deal-towards-sparse-and-robust-pre","slug":"a-win-win-deal-towards-sparse-and-robust-pre","title":"A Win-win Deal: Towards Sparse and Robust Pre-trained Language Models","date":"2022-10-11","arxiv_id":"2210.05211","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["llyx97/sparse-and-robust-plm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-exploration-of-hierarchical-attention","title":"An Exploration of Hierarchical Attention Transformers for Efficient Long Document Classification","date":"2022-10-11","arxiv_id":"2210.05529","n_code_links":0,"syntology":null},{"paper":"/paper/are-pretrained-multilingual-models-equally-1","slug":"are-pretrained-multilingual-models-equally-1","title":"Are Pretrained Multilingual Models Equally Fair Across Languages?","date":"2022-10-11","arxiv_id":"2210.05457","n_code_links":1,"syntology":null},{"paper":null,"slug":"clip-also-understands-text-prompting-clip-for","title":"CLIP also Understands Text: Prompting CLIP for Phrase Understanding","date":"2022-10-11","arxiv_id":"2210.05836","n_code_links":0,"syntology":null},{"paper":"/paper/conserweightive-behavioral-cloning-for","slug":"conserweightive-behavioral-cloning-for","title":"Reliable Conditioning of Behavioral Cloning for Offline Reinforcement Learning","date":"2022-10-11","arxiv_id":"2210.05158","n_code_links":1,"syntology":null},{"paper":"/paper/enriching-biomedical-knowledge-for-low","slug":"enriching-biomedical-knowledge-for-low","title":"Enriching Biomedical Knowledge for Low-resource Language Through Large-Scale Translation","date":"2022-10-11","arxiv_id":"2210.05598","n_code_links":1,"syntology":null},{"paper":null,"slug":"memory-transformers-for-full-context-and-high","title":"Memory transformers for full context and high-resolution 3D Medical Segmentation","date":"2022-10-11","arxiv_id":"2210.05313","n_code_links":0,"syntology":null},{"paper":"/paper/mixture-of-attention-heads-selecting","slug":"mixture-of-attention-heads-selecting","title":"Mixture of Attention Heads: Selecting Attention Heads Per Token","date":"2022-10-11","arxiv_id":"2210.05144","n_code_links":2,"syntology":{"ran":13,"of":17,"n_ran_checked":10,"n_instrument":3,"unverified":4,"pointer_only":4,"phrase":"13 ran (of which 2 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yikangshen/moa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","listed","official"]}}},{"paper":null,"slug":"multilingual-bert-has-an-accent-evaluating","title":"Multilingual BERT has an accent: Evaluating English influences on fluency in multilingual models","date":"2022-10-11","arxiv_id":"2210.05619","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-interpolation-of-contextualized-term","slug":"on-the-interpolation-of-contextualized-term","title":"On the Interpolation of Contextualized Term-based Ranking with BM25 for Query-by-Example Retrieval","date":"2022-10-11","arxiv_id":"2210.05512","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-use-of-semantically-aligned-speech","title":"On the Use of Semantically-Aligned Speech Representations for Spoken Language Understanding","date":"2022-10-11","arxiv_id":"2210.05291","n_code_links":0,"syntology":null},{"paper":"/paper/point-transformer-v2-grouped-vector-attention","slug":"point-transformer-v2-grouped-vector-attention","title":"Point Transformer V2: Grouped Vector Attention and Partition-based Pooling","date":"2022-10-11","arxiv_id":"2210.05666","n_code_links":2,"syntology":null},{"paper":null,"slug":"reflection-of-thought-inversely-eliciting","title":"Reflection of Thought: Inversely Eliciting Numerical Reasoning in Language Models via Solving Linear Systems","date":"2022-10-11","arxiv_id":"2210.05075","n_code_links":0,"syntology":null},{"paper":"/paper/sait-sparse-vision-transformers-through","slug":"sait-sparse-vision-transformers-through","title":"SaiT: Sparse Vision Transformers through Adaptive Token Pruning","date":"2022-10-11","arxiv_id":"2210.05832","n_code_links":1,"syntology":null},{"paper":null,"slug":"streaming-punctuation-for-long-form-dictation","title":"Streaming Punctuation for Long-form Dictation with Transformers","date":"2022-10-11","arxiv_id":"2210.05756","n_code_links":0,"syntology":null},{"paper":"/paper/t5-for-hate-speech-augmented-data-and","slug":"t5-for-hate-speech-augmented-data-and","title":"T5 for Hate Speech, Augmented Data and Ensemble","date":"2022-10-11","arxiv_id":"2210.05480","n_code_links":1,"syntology":null},{"paper":"/paper/understanding-the-failure-of-batch","slug":"understanding-the-failure-of-batch","title":"Understanding the Failure of Batch Normalization for Transformers in NLP","date":"2022-10-11","arxiv_id":"2210.05153","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":8,"n_instrument":1,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wjxts/regularizedbn"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/viterbi-decoding-of-directed-acyclic","slug":"viterbi-decoding-of-directed-acyclic","title":"Viterbi Decoding of Directed Acyclic Transformer for Non-Autoregressive Machine Translation","date":"2022-10-11","arxiv_id":"2210.05193","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-coai/da-transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/vote-n-rank-revision-of-benchmarking-with","slug":"vote-n-rank-revision-of-benchmarking-with","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","date":"2022-10-11","arxiv_id":"2210.05769","n_code_links":1,"syntology":{"ran":6,"of":17,"n_ran_checked":6,"n_instrument":0,"unverified":11,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["pragmaticslab/vote_and_rank"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-memory-transformer-network-for-incremental","title":"A Memory Transformer Network for Incremental Learning","date":"2022-10-10","arxiv_id":"2210.04485","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterization-of-anomalous-diffusion-1","title":"Characterization of anomalous diffusion through convolutional transformers","date":"2022-10-10","arxiv_id":"2210.04959","n_code_links":0,"syntology":null},{"paper":null,"slug":"dcvqe-a-hierarchical-transformer-for-video","title":"DCVQE: A Hierarchical Transformer for Video Quality Assessment","date":"2022-10-10","arxiv_id":"2210.04377","n_code_links":0,"syntology":null},{"paper":"/paper/deptweet-a-typology-for-social-media-texts-to","slug":"deptweet-a-typology-for-social-media-texts-to","title":"DEPTWEET: A Typology for Social Media Texts to Detect Depression Severities","date":"2022-10-10","arxiv_id":"2210.05372","n_code_links":1,"syntology":null},{"paper":"/paper/empowering-the-fact-checkers-automatic","slug":"empowering-the-fact-checkers-automatic","title":"Empowering the Fact-checkers! Automatic Identification of Claim Spans on Twitter","date":"2022-10-10","arxiv_id":"2210.04710","n_code_links":1,"syntology":null},{"paper":"/paper/ensemble-learning-using-transformers-and","slug":"ensemble-learning-using-transformers-and","title":"Ensemble Learning using Transformers and Convolutional Networks for Masked Face Recognition","date":"2022-10-10","arxiv_id":"2210.04816","n_code_links":1,"syntology":null},{"paper":null,"slug":"fs-detr-few-shot-detection-transformer-with","title":"FS-DETR: Few-Shot DEtection TRansformer with prompting and without re-training","date":"2022-10-10","arxiv_id":"2210.04845","n_code_links":0,"syntology":null},{"paper":null,"slug":"lapformer-a-light-and-accurate-polyp","title":"LAPFormer: A Light and Accurate Polyp Segmentation Transformer","date":"2022-10-10","arxiv_id":"2210.04393","n_code_links":0,"syntology":null},{"paper":"/paper/lmqformer-a-laplace-prior-guided-mask-query","slug":"lmqformer-a-laplace-prior-guided-mask-query","title":"LMQFormer: A Laplace-Prior-Guided Mask Query Transformer for Lightweight Snow Removal","date":"2022-10-10","arxiv_id":"2210.04787","n_code_links":1,"syntology":null},{"paper":"/paper/mmt-image-guided-story-ending-generation-with","slug":"mmt-image-guided-story-ending-generation-with","title":"MMT: Image-guided Story Ending Generation with Multimodal Memory Transformer","date":"2022-10-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multi-cls-bert-an-efficient-alternative-to","slug":"multi-cls-bert-an-efficient-alternative-to","title":"Multi-CLS BERT: An Efficient Alternative to Traditional Ensembling","date":"2022-10-10","arxiv_id":"2210.05043","n_code_links":1,"syntology":null},{"paper":"/paper/rev-information-theoretic-evaluation-of-free","slug":"rev-information-theoretic-evaluation-of-free","title":"REV: Information-Theoretic Evaluation of Free-Text Rationales","date":"2022-10-10","arxiv_id":"2210.04982","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hanjiechen/rev"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"revisiting-adapters-with-adversarial-training","title":"Revisiting adapters with adversarial training","date":"2022-10-10","arxiv_id":"2210.04886","n_code_links":0,"syntology":null},{"paper":"/paper/scam-transferring-humans-between-images-with","slug":"scam-transferring-humans-between-images-with","title":"SCAM! Transferring humans between images with Semantic Cross Attention Modulation","date":"2022-10-10","arxiv_id":"2210.04883","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-minimum-wage-as-an-anchor-effects-on","title":"The Minimum Wage as an Anchor: Effects on Determinations of Fairness by Humans and AI","date":"2022-10-10","arxiv_id":"2210.10585","n_code_links":0,"syntology":null},{"paper":"/paper/uncertainty-quantification-with-pre-trained","slug":"uncertainty-quantification-with-pre-trained","title":"Uncertainty Quantification with Pre-trained Language Models: A Large-Scale Empirical Analysis","date":"2022-10-10","arxiv_id":"2210.04714","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xiaoyuxin1002/uq-plm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/visual-prompt-tuning-for-test-time-domain","slug":"visual-prompt-tuning-for-test-time-domain","title":"Visual Prompt Tuning for Test-time Domain Adaptation","date":"2022-10-10","arxiv_id":"2210.04831","n_code_links":0,"syntology":null},{"paper":"/paper/a-transformer-based-deep-neural-network-model","slug":"a-transformer-based-deep-neural-network-model","title":"A Transformer-based deep neural network model for SSVEP classification","date":"2022-10-09","arxiv_id":"2210.04172","n_code_links":2,"syntology":null},{"paper":null,"slug":"ampose-alternatively-mixed-global-local","title":"AMPose: Alternately Mixed Global-Local Attention Model for 3D Human Pose Estimation","date":"2022-10-09","arxiv_id":"2210.04216","n_code_links":0,"syntology":null},{"paper":"/paper/asdot-any-shot-data-to-text-generation-with","slug":"asdot-any-shot-data-to-text-generation-with","title":"ASDOT: Any-Shot Data-to-Text Generation with Pretrained Language Models","date":"2022-10-09","arxiv_id":"2210.04325","n_code_links":1,"syntology":null},{"paper":"/paper/chard-clinical-health-aware-reasoning-across","slug":"chard-clinical-health-aware-reasoning-across","title":"CHARD: Clinical Health-Aware Reasoning Across Dimensions for Text Generation Models","date":"2022-10-09","arxiv_id":"2210.04191","n_code_links":1,"syntology":null},{"paper":"/paper/contra-con-text-tra-nsformer-for-cross-modal","slug":"contra-con-text-tra-nsformer-for-cross-modal","title":"ConTra: (Con)text (Tra)nsformer for Cross-Modal Video Retrieval","date":"2022-10-09","arxiv_id":"2210.04341","n_code_links":1,"syntology":null},{"paper":"/paper/controllable-dialogue-simulation-with-in","slug":"controllable-dialogue-simulation-with-in","title":"Controllable Dialogue Simulation with In-Context Learning","date":"2022-10-09","arxiv_id":"2210.04185","n_code_links":1,"syntology":null},{"paper":"/paper/deep-span-representations-for-named-entity","slug":"deep-span-representations-for-named-entity","title":"Deep Span Representations for Named Entity Recognition","date":"2022-10-09","arxiv_id":"2210.04182","n_code_links":1,"syntology":null},{"paper":"/paper/fairger-using-nlp-to-measure-support-for","slug":"fairger-using-nlp-to-measure-support-for","title":"Fine-Grained Detection of Solidarity for Women and Migrants in 155 Years of German Parliamentary Debates","date":"2022-10-09","arxiv_id":"2210.04359","n_code_links":2,"syntology":null},{"paper":"/paper/fine-tuning-pre-trained-transformers-into","slug":"fine-tuning-pre-trained-transformers-into","title":"Fine-Tuning Pre-trained Transformers into Decaying Fast Weights","date":"2022-10-09","arxiv_id":"2210.04243","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jenni-ai/t2fw"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improve-transformer-pre-training-with","title":"Better Pre-Training by Reducing Representation Confusion","date":"2022-10-09","arxiv_id":"2210.04246","n_code_links":0,"syntology":null},{"paper":null,"slug":"ksat-knowledge-infused-self-attention","title":"KSAT: Knowledge-infused Self Attention Transformer -- Integrating Multiple Domain-Specific Contexts","date":"2022-10-09","arxiv_id":"2210.04307","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-texture-transformer-network-for","title":"Learning Texture Transformer Network for Light Field Super-Resolution","date":"2022-10-09","arxiv_id":"2210.09293","n_code_links":0,"syntology":null},{"paper":"/paper/spread-love-not-hate-undermining-the","slug":"spread-love-not-hate-undermining-the","title":"Spread Love Not Hate: Undermining the Importance of Hateful Pre-training for Hate Speech Detection","date":"2022-10-09","arxiv_id":"2210.04267","n_code_links":1,"syntology":null},{"paper":"/paper/strong-gravitational-lensing-parameter","slug":"strong-gravitational-lensing-parameter","title":"Strong Gravitational Lensing Parameter Estimation with Vision Transformer","date":"2022-10-09","arxiv_id":"2210.04143","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kuanweih/strong_lensing_vit_resnet"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/transformer-based-flood-scene-segmentation","slug":"transformer-based-flood-scene-segmentation","title":"Transformer-based Flood Scene Segmentation for Developing Countries","date":"2022-10-09","arxiv_id":"2210.04218","n_code_links":0,"syntology":null},{"paper":"/paper/volta-vision-language-transformer-with-weakly","slug":"volta-vision-language-transformer-with-weakly","title":"VoLTA: Vision-Language Transformer with Weakly-Supervised Local-Feature Alignment","date":"2022-10-09","arxiv_id":"2210.04135","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":9,"n_instrument":3,"unverified":2,"pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ShramanPramanick/VoLTA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"alphatuning-quantization-aware-parameter","title":"AlphaTuning: Quantization-Aware Parameter-Efficient Adaptation of Large-Scale Pre-Trained Language Models","date":"2022-10-08","arxiv_id":"2210.03858","n_code_links":0,"syntology":null},{"paper":"/paper/fbnet-feedback-network-for-point-cloud","slug":"fbnet-feedback-network-for-point-cloud","title":"FBNet: Feedback Network for Point Cloud Completion","date":"2022-10-08","arxiv_id":"2210.03974","n_code_links":1,"syntology":null},{"paper":"/paper/hierarchical-graph-transformer-with-adaptive","slug":"hierarchical-graph-transformer-with-adaptive","title":"Hierarchical Graph Transformer with Adaptive Node Sampling","date":"2022-10-08","arxiv_id":"2210.03930","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zaixizhang/ans-gt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kg-mtt-bert-knowledge-graph-enhanced-bert-for","title":"KG-MTT-BERT: Knowledge Graph Enhanced BERT for Multi-Type Medical Text Classification","date":"2022-10-08","arxiv_id":"2210.03970","n_code_links":0,"syntology":null}],"record_sha256":"262e39e03620bacb01257e8e2d2f90405fd86a67c4de65817fcd57594521f782","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}