{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/label-smoothing/papers/127","list_of":"/method/label-smoothing","method":"Label Smoothing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":127,"pages_in_order":144,"rows_per_page":100,"rows":[12601,12700],"of":14327,"counts":{"archive_papers_tagged":14327,"with_a_code_link":6651,"where_syntology_ran_a_sample":2259,"not_listed_spam_title":0,"listed":14327,"listed_where_code_ran":2259,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1920,"every_run_a_failure_of_syntologys_instrument":339,"listed_with_a_run_with_no_instrument_failure":1920,"listed_every_run_a_failure_of_syntologys_instrument":339,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/label-smoothing","prev":"/method/label-smoothing/papers/126","next":"/method/label-smoothing/papers/128","papers":[{"paper":"/paper/proof-artifact-co-training-for-theorem","slug":"proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","arxiv_id":"2102.06203","n_code_links":4,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jasonrute/lean-proof-recording-public","jasonrute/lean_proof_recording","jesse-michael-han/lean-step-public"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-compression-aided-transformer-encoding","title":"Text Compression-aided Transformer Encoding","date":"2021-02-11","arxiv_id":"2102.05951","n_code_links":0,"syntology":null},{"paper":"/paper/nast-non-autoregressive-spatial-temporal","slug":"nast-non-autoregressive-spatial-temporal","title":"NAST: Non-Autoregressive Spatial-Temporal Transformer for Time Series Forecasting","date":"2021-02-10","arxiv_id":"2102.05624","n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-transformer-language-models-for","title":"Bayesian Transformer Language Models for Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04754","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-query-rewriting-with-self","title":"Conversational Query Rewriting with Self-supervised Learning","date":"2021-02-09","arxiv_id":"2102.04708","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-intent-detection-and-slot-filling-with","title":"Joint Intent Detection and Slot Filling with Wheel-Graph Attention Networks","date":"2021-02-09","arxiv_id":"2102.04610","n_code_links":0,"syntology":null},{"paper":"/paper/point-cloud-transformers-applied-to-collider","slug":"point-cloud-transformers-applied-to-collider","title":"Point Cloud Transformers applied to Collider Physics","date":"2021-02-09","arxiv_id":"2102.05073","n_code_links":1,"syntology":null},{"paper":"/paper/colorization-transformer-1","slug":"colorization-transformer-1","title":"Colorization Transformer","date":"2021-02-08","arxiv_id":"2102.04432","n_code_links":2,"syntology":null},{"paper":"/paper/transreid-transformer-based-object-re","slug":"transreid-transformer-based-object-re","title":"TransReID: Transformer-based Object Re-Identification","date":"2021-02-08","arxiv_id":"2102.04378","n_code_links":4,"syntology":null},{"paper":"/paper/transunet-transformers-make-strong-encoders","slug":"transunet-transformers-make-strong-encoders","title":"TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation","date":"2021-02-08","arxiv_id":"2102.04306","n_code_links":22,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Beckschen/TransUNet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"wake-word-detection-with-streaming","title":"Wake Word Detection with Streaming Transformers","date":"2021-02-08","arxiv_id":"2102.04488","n_code_links":0,"syntology":null},{"paper":"/paper/nystromformer-a-nystrom-based-algorithm-for","slug":"nystromformer-a-nystrom-based-algorithm-for","title":"Nyströmformer: A Nyström-Based Algorithm for Approximating Self-Attention","date":"2021-02-07","arxiv_id":"2102.03902","n_code_links":10,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlpen/Nystromformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/pipetransformer-automated-elastic-pipelining","slug":"pipetransformer-automated-elastic-pipelining","title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","date":"2021-02-05","arxiv_id":"2102.03161","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Distributed-AI/PipeTransformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mufasa-multimodal-fusion-architecture-search","title":"MUFASA: Multimodal Fusion Architecture Search for Electronic Health Records","date":"2021-02-03","arxiv_id":"2102.02340","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-transfer-learning-with-transformers","title":"Introduction to Neural Transfer Learning with Transformers for Social Science Text Analysis","date":"2021-02-03","arxiv_id":"2102.02111","n_code_links":0,"syntology":null},{"paper":"/paper/pitfalls-of-static-language-modelling","slug":"pitfalls-of-static-language-modelling","title":"Mind the Gap: Assessing Temporal Generalization in Neural Language Models","date":"2021-02-03","arxiv_id":"2102.01951","n_code_links":1,"syntology":null},{"paper":"/paper/relaxed-transformer-decoders-for-direct","slug":"relaxed-transformer-decoders-for-direct","title":"Relaxed Transformer Decoders for Direct Action Proposal Generation","date":"2021-02-03","arxiv_id":"2102.01894","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["MCG-NJU/RTD-Action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-natural-and-controllable-cross","title":"Towards Natural and Controllable Cross-Lingual Voice Conversion Based on Neural TTS Model and Phonetic Posteriorgram","date":"2021-02-03","arxiv_id":"2102.01991","n_code_links":0,"syntology":null},{"paper":"/paper/automated-query-reformulation-for-efficient","slug":"automated-query-reformulation-for-efficient","title":"Automated Query Reformulation for Efficient Search based on Query Logs From Stack Overflow","date":"2021-02-01","arxiv_id":"2102.00826","n_code_links":1,"syntology":null},{"paper":null,"slug":"gtae-graph-transformer-based-auto-encoders","title":"GTAE: Graph-Transformer based Auto-Encoders for Linguistic-Constrained Text Style Transfer","date":"2021-02-01","arxiv_id":"2102.00769","n_code_links":0,"syntology":null},{"paper":"/paper/classification-of-fracture-and-normal","slug":"classification-of-fracture-and-normal","title":"Classification of Shoulder X-Ray Images with Deep Learning Ensemble Models","date":"2021-01-31","arxiv_id":"2102.00515","n_code_links":0,"syntology":null},{"paper":"/paper/computational-performance-predictions-for","slug":"computational-performance-predictions-for","title":"A Runtime-Based Computational Performance Predictor for Deep Neural Network Training","date":"2021-01-31","arxiv_id":"2102.00527","n_code_links":1,"syntology":null},{"paper":"/paper/short-text-clustering-with-transformers","slug":"short-text-clustering-with-transformers","title":"Short Text Clustering with Transformers","date":"2021-01-31","arxiv_id":"2102.00541","n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-graph-decoder-for-neural","slug":"transition-based-graph-decoder-for-neural","title":"Enhancing the Transformer Decoder with Transition-based Syntax","date":"2021-01-29","arxiv_id":"2101.12640","n_code_links":1,"syntology":null},{"paper":null,"slug":"lstm-sakt-lstm-encoded-sakt-like-transformer","title":"LSTM-SAKT: LSTM-Encoded SAKT-like Transformer for Knowledge Tracing","date":"2021-01-28","arxiv_id":"2102.00845","n_code_links":0,"syntology":null},{"paper":"/paper/tokens-to-token-vit-training-vision","slug":"tokens-to-token-vit-training-vision","title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","date":"2021-01-28","arxiv_id":"2101.11986","n_code_links":13,"syntology":{"ran":21,"of":26,"n_ran_checked":21,"n_instrument":0,"unverified":5,"pointer_only":8,"phrase":"21 ran (of which 16 constructed an object rather than computing a result; 21 with no instrument failure: 1 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yitu-opensource/T2T-ViT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"an-explainable-transformer-based-deep","title":"An explainable Transformer-based deep learning model for the prediction of incident heart failure","date":"2021-01-27","arxiv_id":"2101.11359","n_code_links":0,"syntology":null},{"paper":"/paper/bottleneck-transformers-for-visual","slug":"bottleneck-transformers-for-visual","title":"Bottleneck Transformers for Visual Recognition","date":"2021-01-27","arxiv_id":"2101.11605","n_code_links":13,"syntology":{"ran":26,"of":49,"n_ran_checked":19,"n_instrument":7,"unverified":23,"pointer_only":8,"phrase":"26 ran (of which 9 constructed an object rather than computing a result; 19 with no instrument failure: 1 honoured, 0 violated, 18 with no contract checked; 7 where Syntology's instrument failed) · 23 unverified","official":null}},{"paper":"/paper/exploring-multi-task-multi-lingual-learning","slug":"exploring-multi-task-multi-lingual-learning","title":"Exploring multi-task multi-lingual learning of transformer models for hate speech and offensive speech identification in social media","date":"2021-01-27","arxiv_id":"2101.11155","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatial-channel-transformer-network-for","title":"Spatial-Channel Transformer Network for Trajectory Prediction on the Traffic Scenes","date":"2021-01-27","arxiv_id":"2101.11472","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-can-reflect-syntactic-structure-if","title":"Attention Can Reflect Syntactic Structure (If You Let It)","date":"2021-01-26","arxiv_id":"2101.10927","n_code_links":0,"syntology":null},{"paper":null,"slug":"cptr-full-transformer-network-for-image","title":"CPTR: Full Transformer Network for Image Captioning","date":"2021-01-26","arxiv_id":"2101.10804","n_code_links":0,"syntology":null},{"paper":null,"slug":"randomized-deep-structured-prediction-for","title":"Randomized Deep Structured Prediction for Discourse-Level Processing","date":"2021-01-25","arxiv_id":"2101.10435","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-time-series-forecasting-with","title":"Multi-Task Time Series Forecasting With Shared Attention","date":"2021-01-24","arxiv_id":"2101.09645","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-better-integration-of-fuzzy-matches","slug":"towards-a-better-integration-of-fuzzy-matches","title":"Towards a Better Integration of Fuzzy Matches in Neural Machine Translation through Data Augmentation","date":"2021-01-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"enriching-non-autoregressive-transformer-with","title":"Enriching Non-Autoregressive Transformer with Syntactic and SemanticStructures for Neural Machine Translation","date":"2021-01-22","arxiv_id":"2101.08942","n_code_links":0,"syntology":null},{"paper":"/paper/activity-graph-transformer-for-temporal","slug":"activity-graph-transformer-for-temporal","title":"Activity Graph Transformer for Temporal Action Localization","date":"2021-01-21","arxiv_id":"2101.08540","n_code_links":0,"syntology":null},{"paper":"/paper/daf-re-a-challenging-crowd-sourced-large","slug":"daf-re-a-challenging-crowd-sourced-large","title":"DAF:re: A Challenging, Crowd-Sourced, Large-Scale, Long-Tailed Dataset For Anime Character Recognition","date":"2021-01-21","arxiv_id":"2101.08674","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["arkel23/animesion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/evaluating-multilingual-text-encoders-for","slug":"evaluating-multilingual-text-encoders-for","title":"Evaluating Multilingual Text Encoders for Unsupervised Cross-Lingual Retrieval","date":"2021-01-21","arxiv_id":"2101.08370","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rlitschk/EncoderCLIR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learn-to-dance-with-aist-music-conditioned-3d","slug":"learn-to-dance-with-aist-music-conditioned-3d","title":"AI Choreographer: Music Conditioned 3D Dance Generation with AIST++","date":"2021-01-21","arxiv_id":"2101.08779","n_code_links":1,"syntology":null},{"paper":null,"slug":"active-learning-for-sequence-tagging-with","title":"Active Learning for Sequence Tagging with Deep Pre-trained Models and Bayesian Uncertainty Estimates","date":"2021-01-20","arxiv_id":"2101.08133","n_code_links":0,"syntology":null},{"paper":"/paper/open-domain-conversational-search-assistant","slug":"open-domain-conversational-search-assistant","title":"Open-Domain Conversational Search Assistant with Transformers","date":"2021-01-20","arxiv_id":"2101.08197","n_code_links":1,"syntology":null},{"paper":"/paper/pgt-pseudo-relevance-feedback-using-a-graph","slug":"pgt-pseudo-relevance-feedback-using-a-graph","title":"PGT: Pseudo Relevance Feedback Using a Graph-Based Transformer","date":"2021-01-20","arxiv_id":"2101.07918","n_code_links":1,"syntology":null},{"paper":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","n_code_links":1,"syntology":null},{"paper":"/paper/fast-convergence-of-detr-with-spatially","slug":"fast-convergence-of-detr-with-spatially","title":"Fast Convergence of DETR with Spatially Modulated Co-Attention","date":"2021-01-19","arxiv_id":"2101.07448","n_code_links":2,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["gaopengcuhk/SMCA-DETR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/dual-level-collaborative-transformer-for","slug":"dual-level-collaborative-transformer-for","title":"Dual-Level Collaborative Transformer for Image Captioning","date":"2021-01-16","arxiv_id":"2101.06462","n_code_links":1,"syntology":null},{"paper":"/paper/match-ignition-plugging-pagerank-into","slug":"match-ignition-plugging-pagerank-into","title":"Match-Ignition: Plugging PageRank into Transformer for Long-form Text Matching","date":"2021-01-16","arxiv_id":"2101.06423","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploration-of-visual-features-and-their","title":"Exploration of Visual Features and their weighted-additive fusion for Video Captioning","date":"2021-01-14","arxiv_id":"2101.05806","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-analysis-in-language-models-via","title":"Training Data Leakage Analysis in Language Models","date":"2021-01-14","arxiv_id":"2101.05405","n_code_links":0,"syntology":null},{"paper":"/paper/coarse-and-fine-grained-hostility-detection","slug":"coarse-and-fine-grained-hostility-detection","title":"Coarse and Fine-Grained Hostility Detection in Hindi Posts using Fine Tuned Multilingual Embeddings","date":"2021-01-13","arxiv_id":"2101.04998","n_code_links":1,"syntology":null},{"paper":null,"slug":"neural-news-recommendation-with-negative","title":"Neural News Recommendation with Negative Feedback","date":"2021-01-12","arxiv_id":"2101.04328","n_code_links":0,"syntology":null},{"paper":"/paper/bert-gt-cross-sentence-n-ary-relation","slug":"bert-gt-cross-sentence-n-ary-relation","title":"BERT-GT: Cross-sentence n-ary relation extraction with BERT and Graph Transformer","date":"2021-01-11","arxiv_id":"2101.04158","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-the-vision-transformer-model","title":"Investigating the Vision Transformer Model for Image Retrieval Tasks","date":"2021-01-11","arxiv_id":"2101.03771","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-mahalanobis-distance-for","slug":"revisiting-mahalanobis-distance-for","title":"Revisiting Mahalanobis Distance for Transformer-Based Out-of-Domain Detection","date":"2021-01-11","arxiv_id":"2101.03778","n_code_links":1,"syntology":null},{"paper":null,"slug":"spherical-transformer-adapting-spherical","title":"Spherical Transformer: Adapting Spherical Signal to CNNs","date":"2021-01-11","arxiv_id":"2101.03848","n_code_links":0,"syntology":null},{"paper":null,"slug":"channel-boosting-feature-ensemble-for-radar","title":"Channel Boosting Feature Ensemble for Radar-based Object Detection","date":"2021-01-10","arxiv_id":"2101.03531","n_code_links":0,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-with-function","slug":"deep-reinforcement-learning-with-function","title":"Deep Reinforcement Learning with Function Properties in Mean Reversion Strategies","date":"2021-01-09","arxiv_id":"2101.03418","n_code_links":1,"syntology":null},{"paper":"/paper/trankit-a-light-weight-transformer-based","slug":"trankit-a-light-weight-transformer-based","title":"Trankit: A Light-Weight Transformer-based Toolkit for Multilingual Natural Language Processing","date":"2021-01-09","arxiv_id":"2101.03289","n_code_links":1,"syntology":null},{"paper":"/paper/leveraging-multilingual-transformers-for-hate","slug":"leveraging-multilingual-transformers-for-hate","title":"Leveraging Multilingual Transformers for Hate Speech Detection","date":"2021-01-08","arxiv_id":"2101.03207","n_code_links":1,"syntology":null},{"paper":"/paper/compound-word-transformer-learning-to-compose","slug":"compound-word-transformer-learning-to-compose","title":"Compound Word Transformer: Learning to Compose Full-Song Music over Dynamic Directed Hypergraphs","date":"2021-01-07","arxiv_id":"2101.02402","n_code_links":5,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["YatingMusic/compound-word-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"more-reliable-ai-solution-breast-ultrasound","title":"More Reliable AI Solution: Breast Ultrasound Diagnosis Using Multi-AI Combination","date":"2021-01-07","arxiv_id":"2101.02639","n_code_links":0,"syntology":null},{"paper":"/paper/trackformer-multi-object-tracking-with","slug":"trackformer-multi-object-tracking-with","title":"TrackFormer: Multi-Object Tracking with Transformers","date":"2021-01-07","arxiv_id":"2101.02702","n_code_links":2,"syntology":null},{"paper":"/paper/autodropout-learning-dropout-patterns-to","slug":"autodropout-learning-dropout-patterns-to","title":"AutoDropout: Learning Dropout Patterns to Regularize Deep Networks","date":"2021-01-05","arxiv_id":"2101.01761","n_code_links":1,"syntology":null},{"paper":"/paper/i-bert-integer-only-bert-quantization","slug":"i-bert-integer-only-bert-quantization","title":"I-BERT: Integer-only BERT Quantization","date":"2021-01-05","arxiv_id":"2101.01321","n_code_links":7,"syntology":null},{"paper":"/paper/transformers-and-transfer-learning-for","slug":"transformers-and-transfer-learning-for","title":"Improving Portuguese Semantic Role Labeling with Transformers and Transfer Learning","date":"2021-01-04","arxiv_id":"2101.01213","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformers-in-vision-a-survey","title":"Transformers in Vision: A Survey","date":"2021-01-04","arxiv_id":"2101.01169","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-efficient-transformer-decoder-with","title":"An Efficient Transformer Decoder with Compressed Sub-layers","date":"2021-01-03","arxiv_id":"2101.00542","n_code_links":0,"syntology":null},{"paper":null,"slug":"analogical-reasoning-for-visually-grounded-1","title":"Analogical Reasoning for Visually Grounded Compositional Generalization","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ariel-volume-coding-for-sentence-generation-1","title":"AriEL: Volume Coding for Sentence Generation Comparisons","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-is-not-enough-mitigating-the","title":"Attention Is Not Enough: Mitigating the Distribution Discrepancy in Asynchronous Multimodal Sequence Fusion","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"block-skim-transformer-for-efficient-question","title":"Block Skim Transformer for Efficient Question Answering","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"cluster-former-clustering-based-sparse-1","title":"Cluster-Former: Clustering-based Sparse Transformer for Question Answering","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/crackformer-transformer-network-for-fine","slug":"crackformer-transformer-network-for-fine","title":"CrackFormer: Transformer Network for Fine-Grained Crack Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/deep-representational-re-tuning-using","slug":"deep-representational-re-tuning-using","title":"Deep Representational Re-tuning using Contrastive Tension","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"demystifying-loss-functions-for","title":"Demystifying Loss Functions for Classification","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"discovering-human-interactions-with-large","title":"Discovering Human Interactions With Large-Vocabulary Objects via Query and Multi-Scale Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"do-transformers-understand-polynomial","title":"Do Transformers Understand Polynomial Simplification?","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamic-detr-end-to-end-object-detection-with","title":"Dynamic DETR: End-to-End Object Detection With Dynamic Attention","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/event-based-video-reconstruction-using","slug":"event-based-video-reconstruction-using","title":"Event-Based Video Reconstruction Using Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-routing-strategies-for-multilingual","title":"Exploring Routing Strategies for Multilingual Mixture-of-Experts Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"frequency-aware-spatiotemporal-transformers","title":"Frequency-Aware Spatiotemporal Transformers for Video Inpainting Detection","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generalizing-tree-models-for-improving","title":"Generalizing Tree Models for Improving Prediction Accuracy","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"high-performance-discriminative-tracking-with","title":"High-Performance Discriminative Tracking With Transformers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hypergrid-transformers-towards-a-single-model","title":"HyperGrid Transformers: Towards A Single Model for Multiple Tasks","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-harmonization-with-transformer","slug":"image-harmonization-with-transformer","title":"Image Harmonization With Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-generalizability-of-protein","title":"Improving Generalizability of Protein Sequence Models via Data Augmentations","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-machine-translation-by-searching","title":"Improving Machine Translation by Searching Skip Connections Efficiently","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"ketg-a-knowledge-enhanced-text-generation","title":"KETG: A Knowledge Enhanced Text Generation Framework","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-better-structured-representations","title":"Learning Better Structured Representations Using Low-rank Adaptive Label Smoothing","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"long-range-arena-a-benchmark-for-efficient","title":"Long Range Arena : A Benchmark for Efficient Transformers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"memory-representation-in-transformer","title":"Memory Representation in Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-view-3d-reconstruction-with","title":"Multi-View 3D Reconstruction With Transformers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multimodal-co-attention-transformer-for","slug":"multimodal-co-attention-transformer-for","title":"Multimodal Co-Attention Transformer for Survival Prediction in Gigapixel Whole Slide Images","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"non-iterative-parallel-text-generation-via","title":"Non-iterative Parallel Text Generation via Glancing Transformer","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"on-position-embeddings-in-bert","title":"On Position Embeddings in BERT","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"one-size-doesn-t-fit-all-adaptive-label","title":"One Size Doesn't Fit All: Adaptive Label Smoothing","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"parameterization-of-hypercomplex","title":"Parameterization of Hypercomplex Multiplications","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/phrasetransformer-self-attention-using-local","slug":"phrasetransformer-self-attention-using-local","title":"PhraseTransformer: Self-Attention using Local Context for Semantic Parsing","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"post-training-weighted-quantization-of-neural","title":"Post-Training Weighted Quantization of Neural Networks for Language Models","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"predictive-attention-transformer-improving","title":"Predictive Attention Transformer: Improving Transformer with Attention Map Prediction","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"b9caae82629e156b54d4df3b200887e4f864ab384788bc0c5cfcf80f76fc6316","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}