{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/label-smoothing/papers/117","list_of":"/method/label-smoothing","method":"Label Smoothing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":117,"pages_in_order":144,"rows_per_page":100,"rows":[11601,11700],"of":14327,"counts":{"archive_papers_tagged":14327,"with_a_code_link":6651,"where_syntology_ran_a_sample":2259,"not_listed_spam_title":0,"listed":14327,"listed_where_code_ran":2259,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1920,"every_run_a_failure_of_syntologys_instrument":339,"listed_with_a_run_with_no_instrument_failure":1920,"listed_every_run_a_failure_of_syntologys_instrument":339,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/label-smoothing","prev":"/method/label-smoothing/papers/116","next":"/method/label-smoothing/papers/118","papers":[{"paper":"/paper/the-dawn-of-quantum-natural-language","slug":"the-dawn-of-quantum-natural-language","title":"The Dawn of Quantum Natural Language Processing","date":"2021-10-13","arxiv_id":"2110.06510","n_code_links":2,"syntology":null},{"paper":null,"slug":"transform-and-bitstream-domain-image","title":"Transform and Bitstream Domain Image Classification","date":"2021-10-13","arxiv_id":"2110.06740","n_code_links":0,"syntology":null},{"paper":"/paper/yformer-u-net-inspired-transformer-1","slug":"yformer-u-net-inspired-transformer-1","title":"Yformer: U-Net Inspired Transformer Architecture for Far Horizon Time Series Forecasting","date":"2021-10-13","arxiv_id":"2110.08255","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-guided-generative-models-for","title":"Attention-guided Generative Models for Extractive Question Answering","date":"2021-10-12","arxiv_id":"2110.06393","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-learning-for-representation","slug":"contrastive-learning-for-representation","title":"Contrastive Learning for Representation Degeneration Problem in Sequential Recommendation","date":"2021-10-12","arxiv_id":"2110.05730","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["RuihongQiu/DuoRec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/discodvt-generating-long-text-with-discourse","slug":"discodvt-generating-long-text-with-discourse","title":"DiscoDVT: Generating Long Text with Discourse-Aware Discrete Variational Transformer","date":"2021-10-12","arxiv_id":"2110.05999","n_code_links":1,"syntology":null},{"paper":"/paper/hetformer-heterogeneous-transformer-with","slug":"hetformer-heterogeneous-transformer-with","title":"HETFORMER: Heterogeneous Transformer with Sparse Attention for Long-Text Extractive Summarization","date":"2021-10-12","arxiv_id":"2110.06388","n_code_links":1,"syntology":null},{"paper":"/paper/lightseq-accelerated-training-for-transformer","slug":"lightseq-accelerated-training-for-transformer","title":"LightSeq2: Accelerated Training for Transformer-based Models on GPUs","date":"2021-10-12","arxiv_id":"2110.05722","n_code_links":1,"syntology":null},{"paper":"/paper/mention-memory-incorporating-textual-1","slug":"mention-memory-incorporating-textual-1","title":"Mention Memory: incorporating textual knowledge into Transformers through entity mention attention","date":"2021-10-12","arxiv_id":"2110.06176","n_code_links":1,"syntology":null},{"paper":"/paper/relative-molecule-self-attention-transformer-1","slug":"relative-molecule-self-attention-transformer-1","title":"Relative Molecule Self-Attention Transformer","date":"2021-10-12","arxiv_id":"2110.05841","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/rescoring-sequence-to-sequence-models-for","slug":"rescoring-sequence-to-sequence-models-for","title":"Rescoring Sequence-to-Sequence Models for Text Line Recognition with CTC-Prefixes","date":"2021-10-12","arxiv_id":"2110.05909","n_code_links":1,"syntology":null},{"paper":"/paper/satellite-image-semantic-segmentation","slug":"satellite-image-semantic-segmentation","title":"Satellite Image Semantic Segmentation","date":"2021-10-12","arxiv_id":"2110.05812","n_code_links":1,"syntology":null},{"paper":"/paper/starformer-transformer-with-state-action-1","slug":"starformer-transformer-with-state-action-1","title":"StARformer: Transformer with State-Action-Reward Representations for Visual Reinforcement Learning","date":"2021-10-12","arxiv_id":"2110.06206","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["elicassion/StARformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/adaptively-multi-view-and-temporal-fusing","slug":"adaptively-multi-view-and-temporal-fusing","title":"Adaptive Multi-view and Temporal Fusing Transformer for 3D Human Pose Estimation","date":"2021-10-11","arxiv_id":"2110.05092","n_code_links":0,"syntology":null},{"paper":null,"slug":"emds-7-environmental-microorganism-image","title":"EMDS-7: Environmental Microorganism Image Dataset Seventh Version for Multiple Object Detection Evaluation","date":"2021-10-11","arxiv_id":"2110.07723","n_code_links":0,"syntology":null},{"paper":"/paper/instance-based-label-smoothing-for-better","slug":"instance-based-label-smoothing-for-better","title":"Instance-based Label Smoothing For Better Calibrated Classification Networks","date":"2021-10-11","arxiv_id":"2110.05355","n_code_links":1,"syntology":null},{"paper":"/paper/investigating-transfer-learning-capabilities","slug":"investigating-transfer-learning-capabilities","title":"Investigating Transfer Learning Capabilities of Vision Transformers and CNNs by Fine-Tuning a Single Trainable Block","date":"2021-10-11","arxiv_id":"2110.05270","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-view-self-attention-based-transformer","title":"Multi-View Self-Attention Based Transformer for Speaker Recognition","date":"2021-10-11","arxiv_id":"2110.05036","n_code_links":0,"syntology":null},{"paper":null,"slug":"sru-pioneering-fast-recurrence-with-attention","title":"SRU++: Pioneering Fast Recurrence with Attention for Speech Recognition","date":"2021-10-11","arxiv_id":"2110.05571","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-source-separation-via-bayesian","slug":"unsupervised-source-separation-via-bayesian","title":"Unsupervised Source Separation via Bayesian Inference in the Latent Domain","date":"2021-10-11","arxiv_id":"2110.05313","n_code_links":1,"syntology":null},{"paper":null,"slug":"dct-dynamic-compressive-transformer-for","title":"DCT: Dynamic Compressive Transformer for Modeling Unbounded Sequence","date":"2021-10-10","arxiv_id":"2110.04821","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-channel-end-to-end-neural-diarization","title":"Multi-Channel End-to-End Neural Diarization with Distributed Microphones","date":"2021-10-10","arxiv_id":"2110.04694","n_code_links":0,"syntology":null},{"paper":"/paper/nvit-vision-transformer-compression-and-1","slug":"nvit-vision-transformer-compression-and-1","title":"Global Vision Transformer Pruning with Hessian-Aware Saliency","date":"2021-10-10","arxiv_id":"2110.04869","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"on-automatic-text-extractive-summarization","title":"Automatic Text Extractive Summarization Based on Graph and Pre-trained Language Model Attention","date":"2021-10-10","arxiv_id":"2110.04878","n_code_links":0,"syntology":null},{"paper":"/paper/sp-gpt2-semantics-improvement-in-vietnamese","slug":"sp-gpt2-semantics-improvement-in-vietnamese","title":"SP-GPT2: Semantics Improvement in Vietnamese Poetry Generation","date":"2021-10-10","arxiv_id":"2110.15723","n_code_links":2,"syntology":null},{"paper":null,"slug":"supershaper-task-agnostic-super-pre-training","title":"SuperShaper: Task-Agnostic Super Pre-training of BERT Models with Variable Hidden Dimensions","date":"2021-10-10","arxiv_id":"2110.04711","n_code_links":0,"syntology":null},{"paper":"/paper/vector-quantized-image-modeling-with-improved-1","slug":"vector-quantized-image-modeling-with-improved-1","title":"Vector-quantized Image Modeling with Improved VQGAN","date":"2021-10-09","arxiv_id":"2110.04627","n_code_links":5,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"2110-06794","title":"The Layout Generation Algorithm of Graphic Design Based on Transformer-CVAE","date":"2021-10-08","arxiv_id":"2110.06794","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-token-attacks-on-vision","title":"Adversarial Token Attacks on Vision Transformers","date":"2021-10-08","arxiv_id":"2110.04337","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-lgm-leveraging-object-context","title":"Context-LGM: Leveraging Object-Context Relation for Context-Aware Object Recognition","date":"2021-10-08","arxiv_id":"2110.04042","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-learning-for-source-code-with","title":"Towards Learning (Dis)-Similarity of Source Code from Program Contrasts","date":"2021-10-08","arxiv_id":"2110.03868","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-generative-networks-using-gaussian-1","title":"Evaluating generative networks using Gaussian mixtures of image features","date":"2021-10-08","arxiv_id":"2110.05240","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-enhanced-hierarchical-graph","slug":"knowledge-enhanced-hierarchical-graph","title":"Knowledge-Enhanced Hierarchical Graph Transformer Network for Multi-Behavior Recommendation","date":"2021-10-08","arxiv_id":"2110.04000","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-adaptive-control-flow-in","title":"Learning Adaptive Control Flow in Transformers for Improved Systematic Generalization","date":"2021-10-08","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"m6-10t-a-sharing-delinking-paradigm-for","title":"M6-10T: A Sharing-Delinking Paradigm for Efficient Multi-Trillion Parameter Pretraining","date":"2021-10-08","arxiv_id":"2110.03888","n_code_links":0,"syntology":null},{"paper":"/paper/multiplex-behavioral-relation-learning-for","slug":"multiplex-behavioral-relation-learning-for","title":"Multiplex Behavioral Relation Learning for Recommendation via Memory Augmented Transformer Network","date":"2021-10-08","arxiv_id":"2110.04002","n_code_links":1,"syntology":null},{"paper":"/paper/rpt-toward-transferable-model-on","slug":"rpt-toward-transferable-model-on","title":"RPT: Toward Transferable Model on Heterogeneous Researcher Data via Pre-Training","date":"2021-10-08","arxiv_id":"2110.07336","n_code_links":1,"syntology":null},{"paper":"/paper/taming-sparsely-activated-transformer-with","slug":"taming-sparsely-activated-transformer-with","title":"Taming Sparsely Activated Transformer with Stochastic Experts","date":"2021-10-08","arxiv_id":"2110.04260","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/stochastic-mixture-of-experts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vidt-an-efficient-and-effective-fully","slug":"vidt-an-efficient-and-effective-fully","title":"ViDT: An Efficient and Effective Fully Transformer-based Object Detector","date":"2021-10-08","arxiv_id":"2110.03921","n_code_links":1,"syntology":null},{"paper":null,"slug":"attention-is-all-you-need-good-embeddings","title":"Attention is All You Need? Good Embeddings with Statistics are enough:Large Scale Audio Understanding without Transformers/ Convolutions/ BERTs/ Mixers/ Attention/ RNNs or ....","date":"2021-10-07","arxiv_id":"2110.03183","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-supermask-pruning-learning-to","slug":"end-to-end-supermask-pruning-learning-to","title":"End-to-End Supermask Pruning: Learning to Prune Image Captioning Models","date":"2021-10-07","arxiv_id":"2110.03298","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiahuei/sparse-image-captioning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-pre-trained-transformer-for","title":"Generative Pre-Trained Transformer for Cardiac Abnormality Detection","date":"2021-10-07","arxiv_id":"2110.04071","n_code_links":0,"syntology":null},{"paper":"/paper/layer-wise-pruning-of-transformer-attention","slug":"layer-wise-pruning-of-transformer-attention","title":"Layer-wise Pruning of Transformer Attention Heads for Efficient Language Modeling","date":"2021-10-07","arxiv_id":"2110.03252","n_code_links":1,"syntology":null},{"paper":null,"slug":"minimum-word-error-training-for-non","title":"Minimum word error training for non-autoregressive Transformer-based code-switching ASR","date":"2021-10-07","arxiv_id":"2110.03573","n_code_links":0,"syntology":null},{"paper":"/paper/adversarial-robustness-comparison-of-vision","slug":"adversarial-robustness-comparison-of-vision","title":"Adversarial Robustness Comparison of Vision Transformer and MLP-Mixer to CNNs","date":"2021-10-06","arxiv_id":"2110.02797","n_code_links":1,"syntology":null},{"paper":"/paper/anomaly-transformer-time-series-anomaly","slug":"anomaly-transformer-time-series-anomaly","title":"Anomaly Transformer: Time Series Anomaly Detection with Association Discrepancy","date":"2021-10-06","arxiv_id":"2110.02642","n_code_links":3,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thuml/Anomaly-Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamically-decoding-source-domain-knowledge","title":"Dynamically Decoding Source Domain Knowledge for Domain Generalization","date":"2021-10-06","arxiv_id":"2110.03027","n_code_links":0,"syntology":null},{"paper":"/paper/geometric-transformers-for-protein-interface","slug":"geometric-transformers-for-protein-interface","title":"Geometric Transformers for Protein Interface Contact Prediction","date":"2021-10-06","arxiv_id":"2110.02423","n_code_links":2,"syntology":null},{"paper":null,"slug":"how-bpe-affects-memorization-in-transformers","title":"How BPE Affects Memorization in Transformers","date":"2021-10-06","arxiv_id":"2110.02782","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-iteratively-solve-routing","slug":"learning-to-iteratively-solve-routing","title":"Learning to Iteratively Solve Routing Problems with Dual-Aspect Collaborative Transformer","date":"2021-10-06","arxiv_id":"2110.02544","n_code_links":2,"syntology":{"ran":9,"of":16,"n_ran_checked":9,"n_instrument":0,"unverified":7,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["yining043/VRP-DACT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/on-neurons-invariant-to-sentence-structural","slug":"on-neurons-invariant-to-sentence-structural","title":"On Neurons Invariant to Sentence Structural Changes in Neural Machine Translation","date":"2021-10-06","arxiv_id":"2110.03067","n_code_links":1,"syntology":null},{"paper":"/paper/ponet-pooling-network-for-efficient-token","slug":"ponet-pooling-network-for-efficient-token","title":"PoNet: Pooling Network for Efficient Token Mixing in Long Sequences","date":"2021-10-06","arxiv_id":"2110.02442","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lxchtan/ponet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/semantic-prediction-which-one-should-come","slug":"semantic-prediction-which-one-should-come","title":"Semantic Prediction: Which One Should Come First, Recognition or Prediction?","date":"2021-10-06","arxiv_id":"2110.02829","n_code_links":1,"syntology":null},{"paper":"/paper/exploiting-twitter-as-source-of-large-corpora","slug":"exploiting-twitter-as-source-of-large-corpora","title":"Exploiting Twitter as Source of Large Corpora of Weakly Similar Pairs for Semantic Sentence Embeddings","date":"2021-10-05","arxiv_id":"2110.02030","n_code_links":1,"syntology":null},{"paper":"/paper/sicilian-translator-a-recipe-for-low-resource","slug":"sicilian-translator-a-recipe-for-low-resource","title":"Sicilian Translator: A Recipe for Low-Resource NMT","date":"2021-10-05","arxiv_id":"2110.01938","n_code_links":1,"syntology":null},{"paper":"/paper/sound-event-detection-transformer-an-event","slug":"sound-event-detection-transformer-an-event","title":"Sound Event Detection Transformer: An Event-based End-to-End Model for Sound Event Detection","date":"2021-10-05","arxiv_id":"2110.02011","n_code_links":1,"syntology":null},{"paper":"/paper/top-n-equivariant-set-and-graph-generation","slug":"top-n-equivariant-set-and-graph-generation","title":"Top-N: Equivariant set and graph generation without exchangeability","date":"2021-10-05","arxiv_id":"2110.02096","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cvignac/top-n"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/word-acquisition-in-neural-language-models","slug":"word-acquisition-in-neural-language-models","title":"Word Acquisition in Neural Language Models","date":"2021-10-05","arxiv_id":"2110.02406","n_code_links":1,"syntology":null},{"paper":"/paper/3d-transformer-molecular-representation-with","slug":"3d-transformer-molecular-representation-with","title":"Molformer: Motif-based Transformer on 3D Heterogeneous Molecular Graphs","date":"2021-10-04","arxiv_id":"2110.01191","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["smiles724/3d-transformer","smiles724/molformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-free-lunch-from-vit-adaptive-attention","slug":"a-free-lunch-from-vit-adaptive-attention","title":"A free lunch from ViT:Adaptive Attention Multi-scale Fusion Transformer for Fine-grained Visual Recognition","date":"2021-10-04","arxiv_id":"2110.01240","n_code_links":0,"syntology":null},{"paper":"/paper/vtamiq-transformers-for-attention-modulated","slug":"vtamiq-transformers-for-attention-modulated","title":"VTAMIQ: Transformers for Attention Modulated Image Quality Assessment","date":"2021-10-04","arxiv_id":"2110.01655","n_code_links":1,"syntology":null},{"paper":null,"slug":"music-playlist-title-generation-a-machine","title":"Music Playlist Title Generation: A Machine-Translation Approach","date":"2021-10-03","arxiv_id":"2110.07354","n_code_links":0,"syntology":null},{"paper":"/paper/implicit-and-explicit-attention-for-zero-shot","slug":"implicit-and-explicit-attention-for-zero-shot","title":"Implicit and Explicit Attention for Zero-Shot Learning","date":"2021-10-02","arxiv_id":"2110.00860","n_code_links":1,"syntology":null},{"paper":"/paper/proto-program-guided-transformer-for-program","slug":"proto-program-guided-transformer-for-program","title":"ProTo: Program-Guided Transformer for Program-Guided Tasks","date":"2021-10-02","arxiv_id":"2110.00804","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sjtuytc/Neurips21-ProTo-Program-guided-Transformers-for-Program-guided-Tasks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"data-efficient-instance-segmentation-with-a","title":"3rd Place Scheme on Instance Segmentation Track of ICCV 2021 VIPriors Challenges","date":"2021-10-01","arxiv_id":"2110.00242","n_code_links":0,"syntology":null},{"paper":"/paper/geometry-attention-transformer-with-position","slug":"geometry-attention-transformer-with-position","title":"Geometry Attention Transformer with Position-aware LSTMs for Image Captioning","date":"2021-10-01","arxiv_id":"2110.00335","n_code_links":1,"syntology":null},{"paper":null,"slug":"bitcoin-transaction-strategy-construction","title":"Bitcoin Transaction Strategy Construction Based on Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14789","n_code_links":0,"syntology":null},{"paper":"/paper/gt-u-net-a-u-net-like-group-transformer","slug":"gt-u-net-a-u-net-like-group-transformer","title":"GT U-Net: A U-Net Like Group Transformer Network for Tooth Root Segmentation","date":"2021-09-30","arxiv_id":"2109.14813","n_code_links":1,"syntology":null},{"paper":"/paper/inducing-transformer-s-compositional","slug":"inducing-transformer-s-compositional","title":"Inducing Transformer's Compositional Generalization Ability via Auxiliary Sequence Prediction Tasks","date":"2021-09-30","arxiv_id":"2109.15256","n_code_links":1,"syntology":null},{"paper":"/paper/learning-to-predict-trustworthiness-with","slug":"learning-to-predict-trustworthiness-with","title":"Learning to Predict Trustworthiness with Steep Slope Loss","date":"2021-09-30","arxiv_id":"2110.00054","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["luoyan407/predict_trustworthiness"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mobtcast-leveraging-auxiliary-trajectory","title":"MobTCast: Leveraging Auxiliary Trajectory Forecasting for Human Mobility Prediction","date":"2021-09-30","arxiv_id":"2110.01401","n_code_links":0,"syntology":null},{"paper":"/paper/prose2poem-the-blessing-of-transformer-based","slug":"prose2poem-the-blessing-of-transformer-based","title":"Prose2Poem: The Blessing of Transformers in Translating Prose to Persian Poetry","date":"2021-09-30","arxiv_id":"2109.14934","n_code_links":1,"syntology":null},{"paper":"/paper/redesigning-the-transformer-architecture-with","slug":"redesigning-the-transformer-architecture-with","title":"Redesigning the Transformer Architecture with Insights from Multi-particle Dynamical Systems","date":"2021-09-30","arxiv_id":"2109.15142","n_code_links":1,"syntology":null},{"paper":"/paper/scientific-evidence-extraction","slug":"scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","arxiv_id":"2110.00061","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":9,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/table-transformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/syntactic-persistence-in-language-models","slug":"syntactic-persistence-in-language-models","title":"Structural Persistence in Language Models: Priming as a Window into Abstract Language Representations","date":"2021-09-30","arxiv_id":"2109.14989","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-collaborative-attention-adaptive-network","title":"A Collaborative Attention Adaptive Network for Financial Market Forecasting","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/a-dot-product-attention-free-transformer","slug":"a-dot-product-attention-free-transformer","title":"A Dot Product Attention Free Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-control-flow-in-transformers","title":"Adaptive Control Flow in Transformers Improves Systematic Generalization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-label-smoothing-with-self-knowledge","title":"Adaptive Label Smoothing with Self-Knowledge","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-wavelet-transformer-network-for-3d","title":"Adaptive Wavelet Transformer Network for 3D Shape Representation Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-robustness-via-adaptive-label","title":"Adversarial Robustness via Adaptive Label Smoothing","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-investigation-on-hardware-aware-vision","title":"An Investigation on Hardware-Aware Vision Transformer Scaling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"an-object-centric-sensitivity-analysis-of","title":"An object-centric sensitivity analysis of deep learning based instance segmentation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-the-implicit-position-encoding","title":"Analyzing the Implicit Position Encoding Ability of Transformer Decoder","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"are-vision-transformers-robust-to-patch-wise","title":"Are Vision Transformers Robust to Patch-wise Perturbations?","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"associated-learning-an-alternative-to-end-to","title":"Associated Learning: an Alternative to End-to-End Backpropagation that Works on CNN, RNN, and Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-based-interpretability-with-concept","title":"Attention-based Interpretability with Concept Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"audio-lottery-speech-recognition-made-ultra","title":"Audio Lottery: Speech Recognition Made Ultra-Lightweight, Noise-Robust, and Transferable","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"coarformer-transformer-for-large-graph-via","title":"Coarformer: Transformer for large graph via graph coarsening","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"crossformer-transformer-with-alternated-cross","title":"Crossformer: Transformer with Alternated Cross-Layer Guidance","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"d-2-etr-decoder-only-detr-with","title":"D$^2$ETR: Decoder-Only DETR with Computationally Efficient Cross-Scale Attention","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distributional-decision-transformer-for","title":"Distributional Decision Transformer for Hindsight Information Matching","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-point-transformer-for-large-scale","title":"Efficient Point Transformer for Large-scale 3D Scene Understanding","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"equivariant-transformers-for-neural-network","title":"Equivariant Transformers for Neural Network based Molecular Potentials","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ernie-sparse-robust-efficient-transformer","title":"ERNIE-SPARSE: Robust Efficient Transformer Through Hierarchically Unifying Isolated Information","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fairness-in-representation-for-multilingual","title":"Fairness in Representation for Multilingual NLP: Insights from Controlled Experiments on Conditional Language Modeling","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"gental-generative-denoising-skip-gram","title":"GenTAL: Generative Denoising Skip-gram Transformer for Unsupervised Binary Code Similarity Detection","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"geometry-entangled-visual-semantic","title":"Geometry-Entangled Visual Semantic Transformer for Image Captioning","date":"2021-09-29","arxiv_id":"2109.14137","n_code_links":0,"syntology":null},{"paper":null,"slug":"graphix-a-pre-trained-graph-edit-model-for","title":"GRAPHIX: A Pre-trained Graph Edit Model for Automated Program Repair","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"grounding-language-representation-with-visual","title":"Grounding Language Representation with Visual Object Information via Cross Modal Pretraining","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"a9ef0dff6b386cfd657af9deafc11f201aab20269c475304da4f0141ec6785a6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}