{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/residual-connection/papers/207","list_of":"/method/residual-connection","method":"Residual Connection","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":207,"pages_in_order":285,"rows_per_page":100,"rows":[20601,20700],"of":28401,"counts":{"archive_papers_tagged":28401,"with_a_code_link":12847,"where_syntology_ran_a_sample":3897,"not_listed_spam_title":0,"listed":28401,"listed_where_code_ran":3897,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":3291,"every_run_a_failure_of_syntologys_instrument":606,"listed_with_a_run_with_no_instrument_failure":3291,"listed_every_run_a_failure_of_syntologys_instrument":606,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/residual-connection","prev":"/method/residual-connection/papers/206","next":"/method/residual-connection/papers/208","papers":[{"paper":null,"slug":"generative-pre-trained-transformer-for","title":"Generative Pre-Trained Transformer for Cardiac Abnormality Detection","date":"2021-10-07","arxiv_id":"2110.04071","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-growth-at-risk-using-a-multi","title":"Investigating Growth at Risk Using a Multi-country Non-parametric Quantile Factor Model","date":"2021-10-07","arxiv_id":"2110.03411","n_code_links":0,"syntology":null},{"paper":"/paper/layer-wise-pruning-of-transformer-attention","slug":"layer-wise-pruning-of-transformer-attention","title":"Layer-wise Pruning of Transformer Attention Heads for Efficient Language Modeling","date":"2021-10-07","arxiv_id":"2110.03252","n_code_links":1,"syntology":null},{"paper":null,"slug":"minimum-word-error-training-for-non","title":"Minimum word error training for non-autoregressive Transformer-based code-switching ASR","date":"2021-10-07","arxiv_id":"2110.03573","n_code_links":0,"syntology":null},{"paper":"/paper/mixer-tts-non-autoregressive-fast-and-compact","slug":"mixer-tts-non-autoregressive-fast-and-compact","title":"Mixer-TTS: non-autoregressive, fast and compact text-to-speech model conditioned on language model embeddings","date":"2021-10-07","arxiv_id":"2110.03584","n_code_links":1,"syntology":null},{"paper":null,"slug":"universality-of-deep-neural-network-lottery","title":"Universality of Winning Tickets: A Renormalization Group Perspective","date":"2021-10-07","arxiv_id":"2110.03210","n_code_links":0,"syntology":null},{"paper":"/paper/8-bit-optimizers-via-block-wise-quantization","slug":"8-bit-optimizers-via-block-wise-quantization","title":"8-bit Optimizers via Block-wise Quantization","date":"2021-10-06","arxiv_id":"2110.02861","n_code_links":3,"syntology":null},{"paper":"/paper/adversarial-robustness-comparison-of-vision","slug":"adversarial-robustness-comparison-of-vision","title":"Adversarial Robustness Comparison of Vision Transformer and MLP-Mixer to CNNs","date":"2021-10-06","arxiv_id":"2110.02797","n_code_links":1,"syntology":null},{"paper":"/paper/anomaly-transformer-time-series-anomaly","slug":"anomaly-transformer-time-series-anomaly","title":"Anomaly Transformer: Time Series Anomaly Detection with Association Discrepancy","date":"2021-10-06","arxiv_id":"2110.02642","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thuml/Anomaly-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamically-decoding-source-domain-knowledge","title":"Dynamically Decoding Source Domain Knowledge for Domain Generalization","date":"2021-10-06","arxiv_id":"2110.03027","n_code_links":0,"syntology":null},{"paper":"/paper/geometric-transformers-for-protein-interface","slug":"geometric-transformers-for-protein-interface","title":"Geometric Transformers for Protein Interface Contact Prediction","date":"2021-10-06","arxiv_id":"2110.02423","n_code_links":2,"syntology":null},{"paper":"/paper/hire-snn-harnessing-the-inherent-robustness-1","slug":"hire-snn-harnessing-the-inherent-robustness-1","title":"HIRE-SNN: Harnessing the Inherent Robustness of Energy-Efficient Deep Spiking Neural Networks by Training with Crafted Input Noise","date":"2021-10-06","arxiv_id":"2110.11417","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["ksouvik52/hiresnn2021"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"how-bpe-affects-memorization-in-transformers","title":"How BPE Affects Memorization in Transformers","date":"2021-10-06","arxiv_id":"2110.02782","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-iteratively-solve-routing","slug":"learning-to-iteratively-solve-routing","title":"Learning to Iteratively Solve Routing Problems with Dual-Aspect Collaborative Transformer","date":"2021-10-06","arxiv_id":"2110.02544","n_code_links":2,"syntology":{"ran":9,"of":16,"n_ran_checked":9,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["yining043/VRP-DACT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/lidsnet-a-lightweight-on-device-intent","slug":"lidsnet-a-lightweight-on-device-intent","title":"LIDSNet: A Lightweight on-device Intent Detection model using Deep Siamese Network","date":"2021-10-06","arxiv_id":"2110.15717","n_code_links":0,"syntology":null},{"paper":"/paper/nus-ids-at-fincausal-2021-dependency-tree-in","slug":"nus-ids-at-fincausal-2021-dependency-tree-in","title":"NUS-IDS at FinCausal 2021: Dependency Tree in Graph Neural Network for Better Cause-Effect Span Detection","date":"2021-10-06","arxiv_id":"2110.02991","n_code_links":1,"syntology":null},{"paper":"/paper/on-neurons-invariant-to-sentence-structural","slug":"on-neurons-invariant-to-sentence-structural","title":"On Neurons Invariant to Sentence Structural Changes in Neural Machine Translation","date":"2021-10-06","arxiv_id":"2110.03067","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-global-convergence-of-gradient-descent-1","title":"On the Global Convergence of Gradient Descent for multi-layer ResNets in the mean-field regime","date":"2021-10-06","arxiv_id":"2110.02926","n_code_links":0,"syntology":null},{"paper":"/paper/ponet-pooling-network-for-efficient-token","slug":"ponet-pooling-network-for-efficient-token","title":"PoNet: Pooling Network for Efficient Token Mixing in Long Sequences","date":"2021-10-06","arxiv_id":"2110.02442","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lxchtan/ponet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/psg-hasoc-dravidian-codemixfire2021","slug":"psg-hasoc-dravidian-codemixfire2021","title":"Pretrained Transformers for Offensive Language Identification in Tanglish","date":"2021-10-06","arxiv_id":"2110.02852","n_code_links":1,"syntology":null},{"paper":"/paper/semantic-prediction-which-one-should-come","slug":"semantic-prediction-which-one-should-come","title":"Semantic Prediction: Which One Should Come First, Recognition or Prediction?","date":"2021-10-06","arxiv_id":"2110.02829","n_code_links":1,"syntology":null},{"paper":null,"slug":"analyzing-the-impact-of-covid-19-on-economy","title":"Analyzing the Impact of COVID-19 on Economy from the Perspective of Users Reviews","date":"2021-10-05","arxiv_id":"2110.02198","n_code_links":0,"syntology":null},{"paper":null,"slug":"asr-rescoring-and-confidence-estimation-with","title":"ASR Rescoring and Confidence Estimation with ELECTRA","date":"2021-10-05","arxiv_id":"2110.01857","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-subspace-analysing-for-semi-supervised","title":"Deep Subspace analysing for Semi-Supervised multi-label classification of Diabetic Foot Ulcer","date":"2021-10-05","arxiv_id":"2110.01795","n_code_links":0,"syntology":null},{"paper":"/paper/disambiguation-bert-for-n-best-rescoring-in","slug":"disambiguation-bert-for-n-best-rescoring-in","title":"BERT Attends the Conversation: Improving Low-Resource Conversational ASR","date":"2021-10-05","arxiv_id":"2110.02267","n_code_links":1,"syntology":null},{"paper":"/paper/distilhubert-speech-representation-learning","slug":"distilhubert-speech-representation-learning","title":"DistilHuBERT: Speech Representation Learning by Layer-wise Distillation of Hidden-unit BERT","date":"2021-10-05","arxiv_id":"2110.01900","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-twitter-as-source-of-large-corpora","slug":"exploiting-twitter-as-source-of-large-corpora","title":"Exploiting Twitter as Source of Large Corpora of Weakly Similar Pairs for Semantic Sentence Embeddings","date":"2021-10-05","arxiv_id":"2110.02030","n_code_links":1,"syntology":null},{"paper":"/paper/foodchem-a-food-chemical-relation-extraction","slug":"foodchem-a-food-chemical-relation-extraction","title":"FoodChem: A food-chemical relation extraction model","date":"2021-10-05","arxiv_id":"2110.02019","n_code_links":1,"syntology":null},{"paper":null,"slug":"hybrid-classical-quantum-method-for-diabetic","title":"Hybrid Classical-Quantum method for Diabetic Foot Ulcer Classification","date":"2021-10-05","arxiv_id":"2110.02222","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-sense-specific-static-embeddings","title":"Learning Sense-Specific Static Embeddings using Contextualised Word Embeddings as a Proxy","date":"2021-10-05","arxiv_id":"2110.02204","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-the-inductive-bias-of-large","title":"Leveraging the Inductive Bias of Large Language Models for Abstract Textual Reasoning","date":"2021-10-05","arxiv_id":"2110.02370","n_code_links":0,"syntology":null},{"paper":"/paper/mobilevit-light-weight-general-purpose-and","slug":"mobilevit-light-weight-general-purpose-and","title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","date":"2021-10-05","arxiv_id":"2110.02178","n_code_links":31,"syntology":{"ran":53,"of":68,"n_ran_checked":44,"n_instrument":9,"unverified":15,"pointer_only":18,"phrase":"53 ran (of which 31 constructed an object rather than computing a result; 44 with no instrument failure: 1 honoured, 0 violated, 43 with no contract checked; 9 where Syntology's instrument failed) · 15 unverified","official":{"repos":["apple/ml-cvnets"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":"/paper/sicilian-translator-a-recipe-for-low-resource","slug":"sicilian-translator-a-recipe-for-low-resource","title":"Sicilian Translator: A Recipe for Low-Resource NMT","date":"2021-10-05","arxiv_id":"2110.01938","n_code_links":1,"syntology":null},{"paper":"/paper/sound-event-detection-transformer-an-event","slug":"sound-event-detection-transformer-an-event","title":"Sound Event Detection Transformer: An Event-based End-to-End Model for Sound Event Detection","date":"2021-10-05","arxiv_id":"2110.02011","n_code_links":1,"syntology":null},{"paper":"/paper/top-n-equivariant-set-and-graph-generation","slug":"top-n-equivariant-set-and-graph-generation","title":"Top-N: Equivariant set and graph generation without exchangeability","date":"2021-10-05","arxiv_id":"2110.02096","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cvignac/top-n"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ur-iw-hnt-at-germeval-2021-an-ensembling","title":"ur-iw-hnt at GermEval 2021: An Ensembling Strategy with Multiple BERT Models","date":"2021-10-05","arxiv_id":"2110.02042","n_code_links":0,"syntology":null},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":"/paper/3d-transformer-molecular-representation-with","slug":"3d-transformer-molecular-representation-with","title":"Molformer: Motif-based Transformer on 3D Heterogeneous Molecular Graphs","date":"2021-10-04","arxiv_id":"2110.01191","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["smiles724/3d-transformer","smiles724/molformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-free-lunch-from-vit-adaptive-attention","slug":"a-free-lunch-from-vit-adaptive-attention","title":"A free lunch from ViT:Adaptive Attention Multi-scale Fusion Transformer for Fine-grained Visual Recognition","date":"2021-10-04","arxiv_id":"2110.01240","n_code_links":0,"syntology":null},{"paper":"/paper/deepa2-a-modular-framework-for-deep-argument","slug":"deepa2-a-modular-framework-for-deep-argument","title":"DeepA2: A Modular Framework for Deep Argument Analysis with Pretrained Neural Text2Text Language Models","date":"2021-10-04","arxiv_id":"2110.01509","n_code_links":1,"syntology":null},{"paper":"/paper/effectiveness-of-optimization-algorithms-in","slug":"effectiveness-of-optimization-algorithms-in","title":"Effectiveness of Optimization Algorithms in Deep Image Classification","date":"2021-10-04","arxiv_id":"2110.01598","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploiting-pre-trained-asr-models-for","title":"Exploiting Pre-Trained ASR Models for Alzheimer's Disease Recognition Through Spontaneous Speech","date":"2021-10-04","arxiv_id":"2110.01493","n_code_links":0,"syntology":null},{"paper":"/paper/juribert-a-masked-language-model-adaptation","slug":"juribert-a-masked-language-model-adaptation","title":"JuriBERT: A Masked-Language Model Adaptation for French Legal Text","date":"2021-10-04","arxiv_id":"2110.01485","n_code_links":1,"syntology":null},{"paper":null,"slug":"max-and-coincidence-neurons-in-neural","title":"Max and Coincidence Neurons in Neural Networks","date":"2021-10-04","arxiv_id":"2110.01218","n_code_links":0,"syntology":null},{"paper":null,"slug":"perhaps-ptlms-should-go-to-school-a-task-to","title":"Perhaps PTLMs Should Go to School -- A Task to Assess Open Book and Closed Book QA","date":"2021-10-04","arxiv_id":"2110.01552","n_code_links":0,"syntology":null},{"paper":null,"slug":"stochastic-anderson-mixing-for-nonconvex","title":"Stochastic Anderson Mixing for Nonconvex Stochastic Optimization","date":"2021-10-04","arxiv_id":"2110.01543","n_code_links":0,"syntology":null},{"paper":"/paper/vtamiq-transformers-for-attention-modulated","slug":"vtamiq-transformers-for-attention-modulated","title":"VTAMIQ: Transformers for Attention Modulated Image Quality Assessment","date":"2021-10-04","arxiv_id":"2110.01655","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-examples-generation-for-reducing","title":"Adversarial Examples Generation for Reducing Implicit Gender Bias in Pre-trained Models","date":"2021-10-03","arxiv_id":"2110.01094","n_code_links":0,"syntology":null},{"paper":null,"slug":"ear-u-net-efficientnet-and-attention-based","title":"EAR-U-Net: EfficientNet and attention-based residual U-Net for automatic liver segmentation in CT","date":"2021-10-03","arxiv_id":"2110.01014","n_code_links":0,"syntology":null},{"paper":null,"slug":"music-playlist-title-generation-a-machine","title":"Music Playlist Title Generation: A Machine-Translation Approach","date":"2021-10-03","arxiv_id":"2110.07354","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-paradigm-for-information","title":"Unsupervised paradigm for information extraction from transcripts using BERT","date":"2021-10-03","arxiv_id":"2110.00949","n_code_links":0,"syntology":null},{"paper":null,"slug":"artificial-intelligence-for-sustainable","title":"Artificial intelligence for Sustainable Energy: A Contextual Topic Modeling and Content Analysis","date":"2021-10-02","arxiv_id":"2110.00828","n_code_links":0,"syntology":null},{"paper":null,"slug":"explainable-event-recognition","title":"Explainable Event Recognition","date":"2021-10-02","arxiv_id":"2110.00755","n_code_links":0,"syntology":null},{"paper":"/paper/implicit-and-explicit-attention-for-zero-shot","slug":"implicit-and-explicit-attention-for-zero-shot","title":"Implicit and Explicit Attention for Zero-Shot Learning","date":"2021-10-02","arxiv_id":"2110.00860","n_code_links":1,"syntology":null},{"paper":"/paper/proto-program-guided-transformer-for-program","slug":"proto-program-guided-transformer-for-program","title":"ProTo: Program-Guided Transformer for Program-Guided Tasks","date":"2021-10-02","arxiv_id":"2110.00804","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"significance-of-data-augmentation-for","title":"Significance of Data Augmentation for Improving Cleft Lip and Palate Speech Recognition","date":"2021-10-02","arxiv_id":"2110.00797","n_code_links":0,"syntology":null},{"paper":"/paper/swiss-judgment-prediction-a-multilingual","slug":"swiss-judgment-prediction-a-multilingual","title":"Swiss-Judgment-Prediction: A Multilingual Legal Judgment Prediction Benchmark","date":"2021-10-02","arxiv_id":"2110.00806","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert4gcn-using-bert-intermediate-layers-to","title":"BERT4GCN: Using BERT Intermediate Layers to Augment GCN for Aspect-based Sentiment Classification","date":"2021-10-01","arxiv_id":"2110.00171","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-efficient-instance-segmentation-with-a","title":"3rd Place Scheme on Instance Segmentation Track of ICCV 2021 VIPriors Challenges","date":"2021-10-01","arxiv_id":"2110.00242","n_code_links":0,"syntology":null},{"paper":"/paper/geometry-attention-transformer-with-position","slug":"geometry-attention-transformer-with-position","title":"Geometry Attention Transformer with Position-aware LSTMs for Image Captioning","date":"2021-10-01","arxiv_id":"2110.00335","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-punctuation-restoration-for-speech","title":"Improving Punctuation Restoration for Speech Transcripts via External Data","date":"2021-10-01","arxiv_id":"2110.00560","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":"/paper/resnet-strikes-back-an-improved-training","slug":"resnet-strikes-back-an-improved-training","title":"ResNet strikes back: An improved training procedure in timm","date":"2021-10-01","arxiv_id":"2110.00476","n_code_links":14,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["rwightman/pytorch-image-models"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"paper":null,"slug":"span-labeling-approach-for-vietnamese-and","title":"Span Labeling Approach for Vietnamese and Chinese Word Segmentation","date":"2021-10-01","arxiv_id":"2110.00156","n_code_links":0,"syntology":null},{"paper":null,"slug":"unpacking-the-interdependent-systems-of","title":"Unpacking the Interdependent Systems of Discrimination: Ableist Bias in NLP Systems through an Intersectional Lens","date":"2021-10-01","arxiv_id":"2110.00521","n_code_links":0,"syntology":null},{"paper":"/paper/bert-got-a-date-introducing-transformers-to","slug":"bert-got-a-date-introducing-transformers-to","title":"BERT got a Date: Introducing Transformers to Temporal Tagging","date":"2021-09-30","arxiv_id":"2109.14927","n_code_links":1,"syntology":null},{"paper":null,"slug":"bitcoin-transaction-strategy-construction","title":"Bitcoin Transaction Strategy Construction Based on Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14789","n_code_links":0,"syntology":null},{"paper":"/paper/covid-19-fake-news-detection-using","slug":"covid-19-fake-news-detection-using","title":"COVID-19 Fake News Detection Using Bidirectional Encoder Representations from Transformers Based Models","date":"2021-09-30","arxiv_id":"2109.14816","n_code_links":1,"syntology":null},{"paper":"/paper/gt-u-net-a-u-net-like-group-transformer","slug":"gt-u-net-a-u-net-like-group-transformer","title":"GT U-Net: A U-Net Like Group Transformer Network for Tooth Root Segmentation","date":"2021-09-30","arxiv_id":"2109.14813","n_code_links":1,"syntology":null},{"paper":"/paper/impact-of-channel-variation-on-one-class","slug":"impact-of-channel-variation-on-one-class","title":"Impact of Channel Variation on One-Class Learning for Spoof Detection","date":"2021-09-30","arxiv_id":"2109.14900","n_code_links":1,"syntology":null},{"paper":"/paper/inducing-transformer-s-compositional","slug":"inducing-transformer-s-compositional","title":"Inducing Transformer's Compositional Generalization Ability via Auxiliary Sequence Prediction Tasks","date":"2021-09-30","arxiv_id":"2109.15256","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-predict-trustworthiness-with","slug":"learning-to-predict-trustworthiness-with","title":"Learning to Predict Trustworthiness with Steep Slope Loss","date":"2021-09-30","arxiv_id":"2110.00054","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["luoyan407/predict_trustworthiness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mobtcast-leveraging-auxiliary-trajectory","title":"MobTCast: Leveraging Auxiliary Trajectory Forecasting for Human Mobility Prediction","date":"2021-09-30","arxiv_id":"2110.01401","n_code_links":0,"syntology":null},{"paper":"/paper/portaspeech-portable-and-high-quality","slug":"portaspeech-portable-and-high-quality","title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech","date":"2021-09-30","arxiv_id":"2109.15166","n_code_links":4,"syntology":{"ran":12,"of":12,"n_ran_checked":9,"n_instrument":3,"unverified":0,"pointer_only":6,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["natspeech/natspeech"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/prose2poem-the-blessing-of-transformer-based","slug":"prose2poem-the-blessing-of-transformer-based","title":"Prose2Poem: The Blessing of Transformers in Translating Prose to Persian Poetry","date":"2021-09-30","arxiv_id":"2109.14934","n_code_links":1,"syntology":null},{"paper":"/paper/redesigning-the-transformer-architecture-with","slug":"redesigning-the-transformer-architecture-with","title":"Redesigning the Transformer Architecture with Insights from Multi-particle Dynamical Systems","date":"2021-09-30","arxiv_id":"2109.15142","n_code_links":1,"syntology":null},{"paper":"/paper/scientific-evidence-extraction","slug":"scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","arxiv_id":"2110.00061","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/table-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"semi-tensor-product-based-tensordecomposition","title":"Semi-tensor Product-based TensorDecomposition for Neural Network Compression","date":"2021-09-30","arxiv_id":"2109.15200","n_code_links":0,"syntology":null},{"paper":"/paper/syntactic-persistence-in-language-models","slug":"syntactic-persistence-in-language-models","title":"Structural Persistence in Language Models: Priming as a Window into Abstract Language Representations","date":"2021-09-30","arxiv_id":"2109.14989","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-collaborative-attention-adaptive-network","title":"A Collaborative Attention Adaptive Network for Financial Market Forecasting","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-dot-product-attention-free-transformer","slug":"a-dot-product-attention-free-transformer","title":"A Dot Product Attention Free Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-control-flow-in-transformers","title":"Adaptive Control Flow in Transformers Improves Systematic Generalization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-wavelet-transformer-network-for-3d","title":"Adaptive Wavelet Transformer Network for 3D Shape Representation Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-investigation-on-hardware-aware-vision","title":"An Investigation on Hardware-Aware Vision Transformer Scaling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-object-centric-sensitivity-analysis-of","title":"An object-centric sensitivity analysis of deep learning based instance segmentation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-the-implicit-position-encoding","title":"Analyzing the Implicit Position Encoding Ability of Transformer Decoder","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"are-bert-families-zero-shot-learners-a-study","title":"Are BERT Families Zero-Shot Learners? A Study on Their Potential and Limitations","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"are-vision-transformers-robust-to-patch-wise","title":"Are Vision Transformers Robust to Patch-wise Perturbations?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"associated-learning-an-alternative-to-end-to","title":"Associated Learning: an Alternative to End-to-End Backpropagation that Works on CNN, RNN, and Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-based-interpretability-with-concept","title":"Attention-based Interpretability with Concept Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"audio-lottery-speech-recognition-made-ultra","title":"Audio Lottery: Speech Recognition Made Ultra-Lightweight, Noise-Robust, and Transferable","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"automo-mixer-an-automated-multi-objective","title":"AutoMO-Mixer: An automated multi-objective multi-layer perspecton Mixer model for medical image based diagnosis","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/cctrans-simplifying-and-improving-crowd","slug":"cctrans-simplifying-and-improving-crowd","title":"CCTrans: Simplifying and Improving Crowd Counting with Transformer","date":"2021-09-29","arxiv_id":"2109.14483","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"certified-adversarial-robustness-under-the","title":"Certified Adversarial Robustness Under the Bounded Support Set","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"coarformer-transformer-for-large-graph-via","title":"Coarformer: Transformer for large graph via graph coarsening","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"collaborative-storytelling-with-human-actors","title":"Collaborative Storytelling with Human Actors and AI Narrators","date":"2021-09-29","arxiv_id":"2109.14728","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-transformer-based-sequence-to","title":"Compressing Transformer-Based Sequence to Sequence Models With Pre-trained Autoencoders for Text Summarization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"computing-the-average-inter-sample-time-of","title":"Computing the average inter-sample time of event-triggered control using quantitative automata","date":"2021-09-29","arxiv_id":"2109.14391","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-learning-is-just-meta-learning","title":"Contrastive Learning is Just Meta-Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-pre-training-for-zero-shot","title":"Contrastive Pre-training for Zero-Shot Information Retrieval","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"9d99670655f4c0ef409e3890c99d0c13d37b6413d5f943418e9013fa15186cca","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}