{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/layer-normalization/papers/194","list_of":"/method/layer-normalization","method":"Layer Normalization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":194,"pages_in_order":250,"rows_per_page":100,"rows":[19301,19400],"of":24980,"counts":{"archive_papers_tagged":24980,"with_a_code_link":11273,"where_syntology_ran_a_sample":3471,"not_listed_spam_title":0,"listed":24980,"listed_where_code_ran":3471,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2923,"every_run_a_failure_of_syntologys_instrument":548,"listed_with_a_run_with_no_instrument_failure":2923,"listed_every_run_a_failure_of_syntologys_instrument":548,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/layer-normalization","prev":"/method/layer-normalization/papers/193","next":"/method/layer-normalization/papers/195","papers":[{"paper":null,"slug":"2110-06794","title":"The Layout Generation Algorithm of Graphic Design Based on Transformer-CVAE","date":"2021-10-08","arxiv_id":"2110.06794","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-token-attacks-on-vision","title":"Adversarial Token Attacks on Vision Transformers","date":"2021-10-08","arxiv_id":"2110.04337","n_code_links":0,"syntology":null},{"paper":null,"slug":"all-in-one-multi-task-learning-bert-models","title":"ALL-IN-ONE: Multi-Task Learning BERT models for Evaluating Peer Assessments","date":"2021-10-08","arxiv_id":"2110.03895","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-lgm-leveraging-object-context","title":"Context-LGM: Leveraging Object-Context Relation for Context-Aware Object Recognition","date":"2021-10-08","arxiv_id":"2110.04042","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-learning-for-source-code-with","title":"Towards Learning (Dis)-Similarity of Source Code from Program Contrasts","date":"2021-10-08","arxiv_id":"2110.03868","n_code_links":0,"syntology":null},{"paper":"/paper/cross-speaker-emotion-transfer-based-on","slug":"cross-speaker-emotion-transfer-based-on","title":"Cross-speaker Emotion Transfer Based on Speaker Condition Layer Normalization and Semi-Supervised Training in Text-To-Speech","date":"2021-10-08","arxiv_id":"2110.04153","n_code_links":1,"syntology":null},{"paper":null,"slug":"development-of-an-extractive-title-generation","title":"Development of an Extractive Title Generation System Using Titles of Papers of Top Conferences for Intermediate English Students","date":"2021-10-08","arxiv_id":"2110.04204","n_code_links":0,"syntology":null},{"paper":"/paper/hydrasum-disentangling-stylistic-features-in-1","slug":"hydrasum-disentangling-stylistic-features-in-1","title":"HydraSum: Disentangling Stylistic Features in Text Summarization using Multi-Decoder Models","date":"2021-10-08","arxiv_id":"2110.04400","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":3,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["salesforce/hydra-sum"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kg-fid-infusing-knowledge-graph-in-fusion-in-1","title":"KG-FiD: Infusing Knowledge Graph in Fusion-in-Decoder for Open-Domain Question Answering","date":"2021-10-08","arxiv_id":"2110.04330","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-enhanced-hierarchical-graph","slug":"knowledge-enhanced-hierarchical-graph","title":"Knowledge-Enhanced Hierarchical Graph Transformer Network for Multi-Behavior Recommendation","date":"2021-10-08","arxiv_id":"2110.04000","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-adaptive-control-flow-in","title":"Learning Adaptive Control Flow in Transformers for Improved Systematic Generalization","date":"2021-10-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/local-and-global-context-based-pairwise","slug":"local-and-global-context-based-pairwise","title":"Local and Global Context-Based Pairwise Models for Sentence Ordering","date":"2021-10-08","arxiv_id":"2110.04291","n_code_links":1,"syntology":null},{"paper":null,"slug":"m6-10t-a-sharing-delinking-paradigm-for","title":"M6-10T: A Sharing-Delinking Paradigm for Efficient Multi-Trillion Parameter Pretraining","date":"2021-10-08","arxiv_id":"2110.03888","n_code_links":0,"syntology":null},{"paper":"/paper/multiplex-behavioral-relation-learning-for","slug":"multiplex-behavioral-relation-learning-for","title":"Multiplex Behavioral Relation Learning for Recommendation via Memory Augmented Transformer Network","date":"2021-10-08","arxiv_id":"2110.04002","n_code_links":1,"syntology":null},{"paper":"/paper/rpt-toward-transferable-model-on","slug":"rpt-toward-transferable-model-on","title":"RPT: Toward Transferable Model on Heterogeneous Researcher Data via Pre-Training","date":"2021-10-08","arxiv_id":"2110.07336","n_code_links":1,"syntology":null},{"paper":null,"slug":"speeding-up-deep-model-training-by-sharing","title":"Speeding up Deep Model Training by Sharing Weights and Then Unsharing","date":"2021-10-08","arxiv_id":"2110.03848","n_code_links":0,"syntology":null},{"paper":"/paper/taming-sparsely-activated-transformer-with","slug":"taming-sparsely-activated-transformer-with","title":"Taming Sparsely Activated Transformer with Stochastic Experts","date":"2021-10-08","arxiv_id":"2110.04260","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/stochastic-mixture-of-experts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-analysis-and-deep-learning-a-network","title":"Text analysis and deep learning: A network approach","date":"2021-10-08","arxiv_id":"2110.04151","n_code_links":0,"syntology":null},{"paper":"/paper/vidt-an-efficient-and-effective-fully","slug":"vidt-an-efficient-and-effective-fully","title":"ViDT: An Efficient and Effective Fully Transformer-based Object Detector","date":"2021-10-08","arxiv_id":"2110.03921","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-transformer-based","title":"A Comparative Study of Transformer-Based Language Models on Extractive Question Answering","date":"2021-10-07","arxiv_id":"2110.03142","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-is-all-you-need-good-embeddings","title":"Attention is All You Need? Good Embeddings with Statistics are enough:Large Scale Audio Understanding without Transformers/ Convolutions/ BERTs/ Mixers/ Attention/ RNNs or ....","date":"2021-10-07","arxiv_id":"2110.03183","n_code_links":0,"syntology":null},{"paper":"/paper/cross-language-learning-for-entity-matching","slug":"cross-language-learning-for-entity-matching","title":"Cross-Language Learning for Entity Matching","date":"2021-10-07","arxiv_id":"2110.03338","n_code_links":1,"syntology":null},{"paper":"/paper/end-to-end-supermask-pruning-learning-to","slug":"end-to-end-supermask-pruning-learning-to","title":"End-to-End Supermask Pruning: Learning to Prune Image Captioning Models","date":"2021-10-07","arxiv_id":"2110.03298","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiahuei/sparse-image-captioning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-pre-trained-transformer-for","title":"Generative Pre-Trained Transformer for Cardiac Abnormality Detection","date":"2021-10-07","arxiv_id":"2110.04071","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-growth-at-risk-using-a-multi","title":"Investigating Growth at Risk Using a Multi-country Non-parametric Quantile Factor Model","date":"2021-10-07","arxiv_id":"2110.03411","n_code_links":0,"syntology":null},{"paper":"/paper/layer-wise-pruning-of-transformer-attention","slug":"layer-wise-pruning-of-transformer-attention","title":"Layer-wise Pruning of Transformer Attention Heads for Efficient Language Modeling","date":"2021-10-07","arxiv_id":"2110.03252","n_code_links":1,"syntology":null},{"paper":null,"slug":"minimum-word-error-training-for-non","title":"Minimum word error training for non-autoregressive Transformer-based code-switching ASR","date":"2021-10-07","arxiv_id":"2110.03573","n_code_links":0,"syntology":null},{"paper":"/paper/mixer-tts-non-autoregressive-fast-and-compact","slug":"mixer-tts-non-autoregressive-fast-and-compact","title":"Mixer-TTS: non-autoregressive, fast and compact text-to-speech model conditioned on language model embeddings","date":"2021-10-07","arxiv_id":"2110.03584","n_code_links":1,"syntology":null},{"paper":null,"slug":"universality-of-deep-neural-network-lottery","title":"Universality of Winning Tickets: A Renormalization Group Perspective","date":"2021-10-07","arxiv_id":"2110.03210","n_code_links":0,"syntology":null},{"paper":"/paper/8-bit-optimizers-via-block-wise-quantization","slug":"8-bit-optimizers-via-block-wise-quantization","title":"8-bit Optimizers via Block-wise Quantization","date":"2021-10-06","arxiv_id":"2110.02861","n_code_links":3,"syntology":null},{"paper":"/paper/adversarial-robustness-comparison-of-vision","slug":"adversarial-robustness-comparison-of-vision","title":"Adversarial Robustness Comparison of Vision Transformer and MLP-Mixer to CNNs","date":"2021-10-06","arxiv_id":"2110.02797","n_code_links":1,"syntology":null},{"paper":"/paper/anomaly-transformer-time-series-anomaly","slug":"anomaly-transformer-time-series-anomaly","title":"Anomaly Transformer: Time Series Anomaly Detection with Association Discrepancy","date":"2021-10-06","arxiv_id":"2110.02642","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thuml/Anomaly-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamically-decoding-source-domain-knowledge","title":"Dynamically Decoding Source Domain Knowledge for Domain Generalization","date":"2021-10-06","arxiv_id":"2110.03027","n_code_links":0,"syntology":null},{"paper":"/paper/geometric-transformers-for-protein-interface","slug":"geometric-transformers-for-protein-interface","title":"Geometric Transformers for Protein Interface Contact Prediction","date":"2021-10-06","arxiv_id":"2110.02423","n_code_links":2,"syntology":null},{"paper":null,"slug":"how-bpe-affects-memorization-in-transformers","title":"How BPE Affects Memorization in Transformers","date":"2021-10-06","arxiv_id":"2110.02782","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-iteratively-solve-routing","slug":"learning-to-iteratively-solve-routing","title":"Learning to Iteratively Solve Routing Problems with Dual-Aspect Collaborative Transformer","date":"2021-10-06","arxiv_id":"2110.02544","n_code_links":2,"syntology":{"ran":9,"of":16,"n_ran_checked":9,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["yining043/VRP-DACT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/lidsnet-a-lightweight-on-device-intent","slug":"lidsnet-a-lightweight-on-device-intent","title":"LIDSNet: A Lightweight on-device Intent Detection model using Deep Siamese Network","date":"2021-10-06","arxiv_id":"2110.15717","n_code_links":0,"syntology":null},{"paper":"/paper/nus-ids-at-fincausal-2021-dependency-tree-in","slug":"nus-ids-at-fincausal-2021-dependency-tree-in","title":"NUS-IDS at FinCausal 2021: Dependency Tree in Graph Neural Network for Better Cause-Effect Span Detection","date":"2021-10-06","arxiv_id":"2110.02991","n_code_links":1,"syntology":null},{"paper":"/paper/on-neurons-invariant-to-sentence-structural","slug":"on-neurons-invariant-to-sentence-structural","title":"On Neurons Invariant to Sentence Structural Changes in Neural Machine Translation","date":"2021-10-06","arxiv_id":"2110.03067","n_code_links":1,"syntology":null},{"paper":"/paper/ponet-pooling-network-for-efficient-token","slug":"ponet-pooling-network-for-efficient-token","title":"PoNet: Pooling Network for Efficient Token Mixing in Long Sequences","date":"2021-10-06","arxiv_id":"2110.02442","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lxchtan/ponet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/psg-hasoc-dravidian-codemixfire2021","slug":"psg-hasoc-dravidian-codemixfire2021","title":"Pretrained Transformers for Offensive Language Identification in Tanglish","date":"2021-10-06","arxiv_id":"2110.02852","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-prediction-which-one-should-come","slug":"semantic-prediction-which-one-should-come","title":"Semantic Prediction: Which One Should Come First, Recognition or Prediction?","date":"2021-10-06","arxiv_id":"2110.02829","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-the-impact-of-covid-19-on-economy","title":"Analyzing the Impact of COVID-19 on Economy from the Perspective of Users Reviews","date":"2021-10-05","arxiv_id":"2110.02198","n_code_links":0,"syntology":null},{"paper":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","n_code_links":0,"syntology":null},{"paper":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","n_code_links":1,"syntology":null},{"paper":"/paper/distilhubert-speech-representation-learning","slug":"distilhubert-speech-representation-learning","title":"DistilHuBERT: Speech Representation Learning by Layer-wise Distillation of Hidden-unit BERT","date":"2021-10-05","arxiv_id":"2110.01900","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-twitter-as-source-of-large-corpora","slug":"exploiting-twitter-as-source-of-large-corpora","title":"Exploiting Twitter as Source of Large Corpora of Weakly Similar Pairs for Semantic Sentence Embeddings","date":"2021-10-05","arxiv_id":"2110.02030","n_code_links":1,"syntology":null},{"paper":"/paper/foodchem-a-food-chemical-relation-extraction","slug":"foodchem-a-food-chemical-relation-extraction","title":"FoodChem: A food-chemical relation extraction model","date":"2021-10-05","arxiv_id":"2110.02019","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-sense-specific-static-embeddings","title":"Learning Sense-Specific Static Embeddings using Contextualised Word Embeddings as a Proxy","date":"2021-10-05","arxiv_id":"2110.02204","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-the-inductive-bias-of-large","title":"Leveraging the Inductive Bias of Large Language Models for Abstract Textual Reasoning","date":"2021-10-05","arxiv_id":"2110.02370","n_code_links":0,"syntology":null},{"paper":"/paper/mobilevit-light-weight-general-purpose-and","slug":"mobilevit-light-weight-general-purpose-and","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","date":"2021-10-05","arxiv_id":"2110.02178","n_code_links":31,"syntology":{"ran":53,"of":68,"n_ran_checked":44,"n_instrument":9,"unverified":15,"pointer_only":18,"phrase":"53 ran (of which 31 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 9 where Syntology's instrument failed) · 15 unverified","official":{"repos":["apple/ml-cvnets"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/sicilian-translator-a-recipe-for-low-resource","slug":"sicilian-translator-a-recipe-for-low-resource","title":"Sicilian Translator: A Recipe for Low-Resource NMT","date":"2021-10-05","arxiv_id":"2110.01938","n_code_links":1,"syntology":null},{"paper":"/paper/sound-event-detection-transformer-an-event","slug":"sound-event-detection-transformer-an-event","title":"Sound Event Detection Transformer: An Event-based End-to-End Model for Sound Event Detection","date":"2021-10-05","arxiv_id":"2110.02011","n_code_links":1,"syntology":null},{"paper":"/paper/top-n-equivariant-set-and-graph-generation","slug":"top-n-equivariant-set-and-graph-generation","title":"Top-N: Equivariant set and graph generation without exchangeability","date":"2021-10-05","arxiv_id":"2110.02096","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cvignac/top-n"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ur-iw-hnt-at-germeval-2021-an-ensembling","title":"ur-iw-hnt at GermEval 2021: An Ensembling Strategy with Multiple BERT Models","date":"2021-10-05","arxiv_id":"2110.02042","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":"/paper/3d-transformer-molecular-representation-with","slug":"3d-transformer-molecular-representation-with","title":"Molformer: Motif-based Transformer on 3D Heterogeneous Molecular Graphs","date":"2021-10-04","arxiv_id":"2110.01191","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["smiles724/3d-transformer","smiles724/molformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-free-lunch-from-vit-adaptive-attention","slug":"a-free-lunch-from-vit-adaptive-attention","title":"A free lunch from ViT:Adaptive Attention Multi-scale Fusion Transformer for Fine-grained Visual Recognition","date":"2021-10-04","arxiv_id":"2110.01240","n_code_links":0,"syntology":null},{"paper":"/paper/deepa2-a-modular-framework-for-deep-argument","slug":"deepa2-a-modular-framework-for-deep-argument","title":"DeepA2: A Modular Framework for Deep Argument Analysis with Pretrained Neural Text2Text Language Models","date":"2021-10-04","arxiv_id":"2110.01509","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","n_code_links":0,"syntology":null},{"paper":"/paper/juribert-a-masked-language-model-adaptation","slug":"juribert-a-masked-language-model-adaptation","title":"JuriBERT: A Masked-Language Model Adaptation for French Legal Text","date":"2021-10-04","arxiv_id":"2110.01485","n_code_links":1,"syntology":null},{"paper":null,"slug":"perhaps-ptlms-should-go-to-school-a-task-to","title":"Perhaps PTLMs Should Go to School -- A Task to Assess Open Book and Closed Book QA","date":"2021-10-04","arxiv_id":"2110.01552","n_code_links":0,"syntology":null},{"paper":"/paper/vtamiq-transformers-for-attention-modulated","slug":"vtamiq-transformers-for-attention-modulated","title":"VTAMIQ: Transformers for Attention Modulated Image Quality Assessment","date":"2021-10-04","arxiv_id":"2110.01655","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-examples-generation-for-reducing","title":"Adversarial Examples Generation for Reducing Implicit Gender Bias in Pre-trained Models","date":"2021-10-03","arxiv_id":"2110.01094","n_code_links":0,"syntology":null},{"paper":null,"slug":"music-playlist-title-generation-a-machine","title":"Music Playlist Title Generation: A Machine-Translation Approach","date":"2021-10-03","arxiv_id":"2110.07354","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-paradigm-for-information","title":"Unsupervised paradigm for information extraction from transcripts using BERT","date":"2021-10-03","arxiv_id":"2110.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-for-sustainable","title":"Artificial intelligence for Sustainable Energy: A Contextual Topic Modeling and Content Analysis","date":"2021-10-02","arxiv_id":"2110.00828","n_code_links":0,"syntology":null},{"paper":"/paper/implicit-and-explicit-attention-for-zero-shot","slug":"implicit-and-explicit-attention-for-zero-shot","title":"Implicit and Explicit Attention for Zero-Shot Learning","date":"2021-10-02","arxiv_id":"2110.00860","n_code_links":1,"syntology":null},{"paper":"/paper/proto-program-guided-transformer-for-program","slug":"proto-program-guided-transformer-for-program","title":"ProTo: Program-Guided Transformer for Program-Guided Tasks","date":"2021-10-02","arxiv_id":"2110.00804","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/swiss-judgment-prediction-a-multilingual","slug":"swiss-judgment-prediction-a-multilingual","title":"Swiss-Judgment-Prediction: A Multilingual Legal Judgment Prediction Benchmark","date":"2021-10-02","arxiv_id":"2110.00806","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert4gcn-using-bert-intermediate-layers-to","title":"BERT4GCN: Using BERT Intermediate Layers to Augment GCN for Aspect-based Sentiment Classification","date":"2021-10-01","arxiv_id":"2110.00171","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-efficient-instance-segmentation-with-a","title":"3rd Place Scheme on Instance Segmentation Track of ICCV 2021 VIPriors Challenges","date":"2021-10-01","arxiv_id":"2110.00242","n_code_links":0,"syntology":null},{"paper":"/paper/geometry-attention-transformer-with-position","slug":"geometry-attention-transformer-with-position","title":"Geometry Attention Transformer with Position-aware LSTMs for Image Captioning","date":"2021-10-01","arxiv_id":"2110.00335","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"span-labeling-approach-for-vietnamese-and","title":"Span Labeling Approach for Vietnamese and Chinese Word Segmentation","date":"2021-10-01","arxiv_id":"2110.00156","n_code_links":0,"syntology":null},{"paper":null,"slug":"unpacking-the-interdependent-systems-of","title":"Unpacking the Interdependent Systems of Discrimination: Ableist Bias in NLP Systems through an Intersectional Lens","date":"2021-10-01","arxiv_id":"2110.00521","n_code_links":0,"syntology":null},{"paper":"/paper/bert-got-a-date-introducing-transformers-to","slug":"bert-got-a-date-introducing-transformers-to","title":"BERT got a Date: Introducing Transformers to Temporal Tagging","date":"2021-09-30","arxiv_id":"2109.14927","n_code_links":1,"syntology":null},{"paper":null,"slug":"bitcoin-transaction-strategy-construction","title":"Bitcoin Transaction Strategy Construction Based on Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14789","n_code_links":0,"syntology":null},{"paper":"/paper/covid-19-fake-news-detection-using","slug":"covid-19-fake-news-detection-using","title":"COVID-19 Fake News Detection Using Bidirectional Encoder Representations from Transformers Based Models","date":"2021-09-30","arxiv_id":"2109.14816","n_code_links":1,"syntology":null},{"paper":"/paper/gt-u-net-a-u-net-like-group-transformer","slug":"gt-u-net-a-u-net-like-group-transformer","title":"GT U-Net: A U-Net Like Group Transformer Network for Tooth Root Segmentation","date":"2021-09-30","arxiv_id":"2109.14813","n_code_links":1,"syntology":null},{"paper":"/paper/inducing-transformer-s-compositional","slug":"inducing-transformer-s-compositional","title":"Inducing Transformer's Compositional Generalization Ability via Auxiliary Sequence Prediction Tasks","date":"2021-09-30","arxiv_id":"2109.15256","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-predict-trustworthiness-with","slug":"learning-to-predict-trustworthiness-with","title":"Learning to Predict Trustworthiness with Steep Slope Loss","date":"2021-09-30","arxiv_id":"2110.00054","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["luoyan407/predict_trustworthiness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mobtcast-leveraging-auxiliary-trajectory","title":"MobTCast: Leveraging Auxiliary Trajectory Forecasting for Human Mobility Prediction","date":"2021-09-30","arxiv_id":"2110.01401","n_code_links":0,"syntology":null},{"paper":"/paper/portaspeech-portable-and-high-quality","slug":"portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","n_code_links":4,"syntology":{"ran":12,"of":12,"n_ran_checked":9,"n_instrument":3,"unverified":0,"pointer_only":6,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["natspeech/natspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/prose2poem-the-blessing-of-transformer-based","slug":"prose2poem-the-blessing-of-transformer-based","title":"Prose2Poem: The Blessing of Transformers in Translating Prose to Persian Poetry","date":"2021-09-30","arxiv_id":"2109.14934","n_code_links":1,"syntology":null},{"paper":"/paper/redesigning-the-transformer-architecture-with","slug":"redesigning-the-transformer-architecture-with","title":"Redesigning the Transformer Architecture with Insights from Multi-particle Dynamical Systems","date":"2021-09-30","arxiv_id":"2109.15142","n_code_links":1,"syntology":null},{"paper":"/paper/scientific-evidence-extraction","slug":"scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","arxiv_id":"2110.00061","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/table-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/syntactic-persistence-in-language-models","slug":"syntactic-persistence-in-language-models","title":"Structural Persistence in Language Models: Priming as a Window into Abstract Language Representations","date":"2021-09-30","arxiv_id":"2109.14989","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-collaborative-attention-adaptive-network","title":"A Collaborative Attention Adaptive Network for Financial Market Forecasting","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-dot-product-attention-free-transformer","slug":"a-dot-product-attention-free-transformer","title":"A Dot Product Attention Free Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-control-flow-in-transformers","title":"Adaptive Control Flow in Transformers Improves Systematic Generalization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-wavelet-transformer-network-for-3d","title":"Adaptive Wavelet Transformer Network for 3D Shape Representation Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-investigation-on-hardware-aware-vision","title":"An Investigation on Hardware-Aware Vision Transformer Scaling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-object-centric-sensitivity-analysis-of","title":"An object-centric sensitivity analysis of deep learning based instance segmentation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-the-implicit-position-encoding","title":"Analyzing the Implicit Position Encoding Ability of Transformer Decoder","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"are-bert-families-zero-shot-learners-a-study","title":"Are BERT Families Zero-Shot Learners? A Study on Their Potential and Limitations","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"are-vision-transformers-robust-to-patch-wise","title":"Are Vision Transformers Robust to Patch-wise Perturbations?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"associated-learning-an-alternative-to-end-to","title":"Associated Learning: an Alternative to End-to-End Backpropagation that Works on CNN, RNN, and Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-based-interpretability-with-concept","title":"Attention-based Interpretability with Concept Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"cf1e9fc9556bb3935448c16f3041e20654cbf8cb7fc488208de5324259f7ed02","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}