{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/weight-decay/papers/63","list_of":"/method/weight-decay","method":"Weight Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":63,"pages_in_order":108,"rows_per_page":100,"rows":[6201,6300],"of":10713,"counts":{"archive_papers_tagged":10713,"with_a_code_link":4533,"where_syntology_ran_a_sample":1291,"not_listed_spam_title":0,"listed":10713,"listed_where_code_ran":1291,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1064,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1064,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/weight-decay","prev":"/method/weight-decay/papers/62","next":"/method/weight-decay/papers/64","papers":[{"paper":null,"slug":"a-survey-on-masked-autoencoder-for-self","title":"A Survey on Masked Autoencoder for Self-supervised Learning in Vision and Beyond","date":"2022-07-30","arxiv_id":"2208.00173","n_code_links":0,"syntology":null},{"paper":"/paper/code-comment-inconsistency-detection-with","slug":"code-comment-inconsistency-detection-with","title":"Code Comment Inconsistency Detection with BERT and Longformer","date":"2022-07-29","arxiv_id":"2207.14444","n_code_links":1,"syntology":null},{"paper":null,"slug":"curriculum-learning-for-data-efficient-vision","title":"Curriculum Learning for Data-Efficient Vision-Language Alignment","date":"2022-07-29","arxiv_id":"2207.14525","n_code_links":0,"syntology":null},{"paper":null,"slug":"sercnn-stacked-embedding-recurrent-1","title":"SERCNN: Stacked Embedding Recurrent Convolutional Neural Network in Detecting Depression on Twitter","date":"2022-07-29","arxiv_id":"2207.14535","n_code_links":0,"syntology":null},{"paper":"/paper/cram-a-compression-aware-minimizer","slug":"cram-a-compression-aware-minimizer","title":"CrAM: A Compression-Aware Minimizer","date":"2022-07-28","arxiv_id":"2207.14200","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["ist-daslab/cram"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lad-language-models-as-data-for-zero-shot","title":"LAD: Language Models as Data for Zero-Shot Dialog","date":"2022-07-28","arxiv_id":"2207.14393","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-and-the-reverse-turing","title":"Large Language Models and the Reverse Turing Test","date":"2022-07-28","arxiv_id":"2207.14382","n_code_links":0,"syntology":null},{"paper":null,"slug":"sdbert-sparsedistilbert-a-faster-and-smaller","title":"SDBERT: SparseDistilBERT, a faster and smaller BERT model","date":"2022-07-28","arxiv_id":"2208.10246","n_code_links":0,"syntology":null},{"paper":"/paper/sequence-to-sequence-pretraining-for-a-less","slug":"sequence-to-sequence-pretraining-for-a-less","title":"Sequence to sequence pretraining for a less-resourced Slovenian language","date":"2022-07-28","arxiv_id":"2207.13988","n_code_links":1,"syntology":null},{"paper":null,"slug":"distributional-actor-critic-ensemble-for","title":"Distributional Actor-Critic Ensemble for Uncertainty-Aware Continuous Control","date":"2022-07-27","arxiv_id":"2207.13730","n_code_links":0,"syntology":null},{"paper":"/paper/soundchoice-grapheme-to-phoneme-models-with","slug":"soundchoice-grapheme-to-phoneme-models-with","title":"SoundChoice: Grapheme-to-Phoneme Models with Semantic Disambiguation","date":"2022-07-27","arxiv_id":"2207.13703","n_code_links":1,"syntology":null},{"paper":"/paper/bundle-mcr-towards-conversational-bundle","slug":"bundle-mcr-towards-conversational-bundle","title":"Bundle MCR: Towards Conversational Bundle Recommendation","date":"2022-07-26","arxiv_id":"2207.12628","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-bert-for-automatic-adme-semantic","title":"Fine-Tuning BERT for Automatic ADME Semantic Labeling in FDA Drug Labeling to Enhance Product-Specific Guidance Assessment","date":"2022-07-25","arxiv_id":"2207.12376","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-gpt-3-all-you-need-for-visual-question","title":"Is GPT-3 all you need for Visual Question Answering in Cultural Heritage?","date":"2022-07-25","arxiv_id":"2207.12101","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-cognitive-study-on-semantic-similarity","title":"A Cognitive Study on Semantic Similarity Analysis of Large Corpora: A Transformer-based Approach","date":"2022-07-24","arxiv_id":"2207.11716","n_code_links":0,"syntology":null},{"paper":null,"slug":"better-reasoning-behind-classification","title":"Better Reasoning Behind Classification Predictions with BERT for Fake News Detection","date":"2022-07-23","arxiv_id":"2207.11562","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-video-captioning-with-evolving","slug":"zero-shot-video-captioning-with-evolving","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","date":"2022-07-22","arxiv_id":"2207.11100","n_code_links":1,"syntology":null},{"paper":null,"slug":"bigissue-a-realistic-bug-localization","title":"BigIssue: A Realistic Bug Localization Benchmark","date":"2022-07-21","arxiv_id":"2207.10739","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-model-compression-with-random","slug":"efficient-model-compression-with-random","title":"Efficient model compression with Random Operation Access Specific Tile (ROAST) hashing","date":"2022-07-21","arxiv_id":"2207.10702","n_code_links":1,"syntology":null},{"paper":"/paper/abstract-demonstrations-and-adaptive","slug":"abstract-demonstrations-and-adaptive","title":"Abstract Demonstrations and Adaptive Exploration for Efficient and Stable Multi-step Sparse Reward Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09243","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-collaborative-filtering-recommender","slug":"enhancing-collaborative-filtering-recommender","title":"Enhancing Collaborative Filtering Recommender with Prompt-Based Sentiment Analysis","date":"2022-07-19","arxiv_id":"2207.12883","n_code_links":1,"syntology":null},{"paper":"/paper/pic-a-phrase-in-context-dataset-for-phrase","slug":"pic-a-phrase-in-context-dataset-for-phrase","title":"PiC: A Phrase-in-Context Dataset for Phrase Understanding and Semantic Search","date":"2022-07-19","arxiv_id":"2207.09068","n_code_links":1,"syntology":null},{"paper":"/paper/pre-trained-language-models-with-domain","slug":"pre-trained-language-models-with-domain","title":"Pre-trained language models with domain knowledge for biomedical extractive summarization","date":"2022-07-19","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"revealing-secrets-from-pre-trained-models","title":"Revealing Secrets From Pre-trained Models","date":"2022-07-19","arxiv_id":"2207.09539","n_code_links":0,"syntology":null},{"paper":"/paper/selection-bias-induced-spurious-correlations","slug":"selection-bias-induced-spurious-correlations","title":"Selection Bias Induced Spurious Correlations in Large Language Models","date":"2022-07-18","arxiv_id":"2207.08982","n_code_links":1,"syntology":null},{"paper":null,"slug":"word-play-for-playing-othello-reverses","title":"Word Play for Playing Othello (Reverses)","date":"2022-07-18","arxiv_id":"2207.08766","n_code_links":0,"syntology":null},{"paper":"/paper/aspect-specific-context-modeling-for-aspect","slug":"aspect-specific-context-modeling-for-aspect","title":"Aspect-specific Context Modeling for Aspect-based Sentiment Analysis","date":"2022-07-17","arxiv_id":"2207.08099","n_code_links":1,"syntology":null},{"paper":"/paper/can-large-language-models-reason-about","slug":"can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","arxiv_id":"2207.08143","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vlievin/medical-reasoning"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/electra-is-a-zero-shot-learner-too","slug":"electra-is-a-zero-shot-learner-too","title":"ELECTRA is a Zero-Shot Learner, Too","date":"2022-07-17","arxiv_id":"2207.08141","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nishiwen1214/rtd-electra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"representation-learning-of-image-schema","title":"Representation Learning of Image Schema","date":"2022-07-17","arxiv_id":"2207.08256","n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-action-governor-for-uncertain","title":"Robust Action Governor for Uncertain Piecewise Affine Systems with Non-convex Constraints and Safe Reinforcement Learning","date":"2022-07-17","arxiv_id":"2207.08240","n_code_links":0,"syntology":null},{"paper":null,"slug":"troll-tweet-detection-using-contextualized","title":"A Context-Sensitive Word Embedding Approach for The Detection of Troll Tweets","date":"2022-07-17","arxiv_id":"2207.08230","n_code_links":0,"syntology":null},{"paper":"/paper/poet-training-neural-networks-on-tiny-devices","slug":"poet-training-neural-networks-on-tiny-devices","title":"POET: Training Neural Networks on Tiny Devices with Integrated Rematerialization and Paging","date":"2022-07-15","arxiv_id":"2207.07697","n_code_links":1,"syntology":null},{"paper":"/paper/position-prediction-as-an-effective","slug":"position-prediction-as-an-effective","title":"Position Prediction as an Effective Pretraining Strategy","date":"2022-07-15","arxiv_id":"2207.07611","n_code_links":1,"syntology":null},{"paper":null,"slug":"z-index-at-checkthat-lab-2022-check","title":"Z-Index at CheckThat! Lab 2022: Check-Worthiness Identification on Tweet Text","date":"2022-07-15","arxiv_id":"2207.07308","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-data-pattern-extraction-attacks-on","title":"Combing for Credentials: Active Pattern Extraction from Smart Reply","date":"2022-07-14","arxiv_id":"2207.10802","n_code_links":0,"syntology":null},{"paper":"/paper/bootstrapped-masked-autoencoders-for-vision","slug":"bootstrapped-masked-autoencoders-for-vision","title":"Bootstrapped Masked Autoencoders for Vision BERT Pretraining","date":"2022-07-14","arxiv_id":"2207.07116","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","official":{"repos":["lightdxy/bootmae"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-modelling-with-pixels","slug":"language-modelling-with-pixels","title":"Language Modelling with Pixels","date":"2022-07-14","arxiv_id":"2207.06991","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xplip/pixel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilinguals-at-semeval-2022-task-11-complex-1","slug":"multilinguals-at-semeval-2022-task-11-complex-1","title":"Multilinguals at SemEval-2022 Task 11: Complex NER in Semantically Ambiguous Settings for Low Resource Languages","date":"2022-07-14","arxiv_id":"2207.06882","n_code_links":1,"syntology":null},{"paper":"/paper/piat-physics-informed-adversarial-training","slug":"piat-physics-informed-adversarial-training","title":"PIAT: Physics Informed Adversarial Training for Solving Partial Differential Equations","date":"2022-07-14","arxiv_id":"2207.06647","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-transfer-learning-based-model-for-text","title":"A Transfer Learning Based Model for Text Readability Assessment in German","date":"2022-07-13","arxiv_id":"2207.06265","n_code_links":0,"syntology":null},{"paper":"/paper/dynast-dynamic-sparse-transformer-for","slug":"dynast-dynamic-sparse-transformer-for","title":"DynaST: Dynamic Sparse Transformer for Exemplar-Guided Image Generation","date":"2022-07-13","arxiv_id":"2207.06124","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":3,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["huage001/dynast"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploiting-word-semantics-to-enrich-character","slug":"exploiting-word-semantics-to-enrich-character","title":"Exploiting Word Semantics to Enrich Character Representations of Chinese Pre-trained Models","date":"2022-07-13","arxiv_id":"2207.05928","n_code_links":1,"syntology":null},{"paper":"/paper/re2g-retrieve-rerank-generate-2","slug":"re2g-retrieve-rerank-generate-2","title":"Re2G: Retrieve, Rerank, Generate","date":"2022-07-13","arxiv_id":"2207.06300","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ibm/kgi-slot-filling"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-do-multilingual-encoders-learn-cross","title":"How Do Multilingual Encoders Learn Cross-lingual Representation?","date":"2022-07-12","arxiv_id":"2207.05737","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-paraphrases-to-study-properties-of-1","title":"Using Paraphrases to Study Properties of Contextual Embeddings","date":"2022-07-12","arxiv_id":"2207.05553","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-large-scale-universal-user","title":"Learning Large-scale Universal User Representation with Sparse Mixture of Experts","date":"2022-07-11","arxiv_id":"2207.04648","n_code_links":0,"syntology":null},{"paper":"/paper/multi-level-fusion-of-wav2vec-2-0-and-bert","slug":"multi-level-fusion-of-wav2vec-2-0-and-bert","title":"Multi-level Fusion of Wav2vec 2.0 and BERT for Multimodal Emotion Recognition","date":"2022-07-11","arxiv_id":"2207.04697","n_code_links":1,"syntology":null},{"paper":null,"slug":"overview-of-the-shared-task-on-fake-news","title":"Overview of the Shared Task on Fake News Detection in Urdu at FIRE 2021","date":"2022-07-11","arxiv_id":"2207.05133","n_code_links":0,"syntology":null},{"paper":"/paper/sparsetir-composable-abstractions-for-sparse","slug":"sparsetir-composable-abstractions-for-sparse","title":"SparseTIR: Composable Abstractions for Sparse Compilation in Deep Learning","date":"2022-07-11","arxiv_id":"2207.04606","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uwsampl/sparsetir","uwsampl/sparsetir-artifact"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilingual-persuasion-detection-video-games","slug":"multilingual-persuasion-detection-video-games","title":"Multilingual Persuasion Detection: Video Games as an Invaluable Data Source for NLP","date":"2022-07-10","arxiv_id":"2207.04453","n_code_links":1,"syntology":null},{"paper":null,"slug":"few-shot-training-llms-for-project-specific","title":"Few-shot training LLMs for project-specific code-summarization","date":"2022-07-09","arxiv_id":"2207.04237","n_code_links":0,"syntology":null},{"paper":"/paper/computationally-identifying-funneling-and-1","slug":"computationally-identifying-funneling-and-1","title":"Computationally Identifying Funneling and Focusing Questions in Classroom Discourse","date":"2022-07-08","arxiv_id":"2208.04715","n_code_links":1,"syntology":null},{"paper":"/paper/deep-visual-linguistic-fusion-network","slug":"deep-visual-linguistic-fusion-network","title":"Deep Visual-Linguistic Fusion Network Considering Cross-Modal Inconsistency for Rumor Detection","date":"2022-07-08","arxiv_id":null,"n_code_links":3,"syntology":null},{"paper":null,"slug":"hidden-schema-networks","title":"Hidden Schema Networks","date":"2022-07-08","arxiv_id":"2207.03777","n_code_links":0,"syntology":null},{"paper":"/paper/a-large-scale-search-dataset-for-unbiased","slug":"a-large-scale-search-dataset-for-unbiased","title":"A Large Scale Search Dataset for Unbiased Learning to Rank","date":"2022-07-07","arxiv_id":"2207.03051","n_code_links":1,"syntology":null},{"paper":null,"slug":"active-learning-and-multi-label","title":"Active Learning and Multi-label Classification for Ellipsis and Coreference Detection in Conversational Question-Answering","date":"2022-07-07","arxiv_id":"2207.03145","n_code_links":0,"syntology":null},{"paper":null,"slug":"asner-annotated-dataset-and-baseline-for","title":"AsNER -- Annotated Dataset and Baseline for Assamese Named Entity recognition","date":"2022-07-07","arxiv_id":"2207.03422","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-language-models-are-not-born-equal-to","title":"Neural Language Models are not Born Equal to Fit Brain Data, but Training Helps","date":"2022-07-07","arxiv_id":"2207.03380","n_code_links":0,"syntology":null},{"paper":null,"slug":"sensitivity-analysis-on-transferred-neural","title":"Sensitivity Analysis on Transferred Neural Architectures of BERT and GPT-2 for Financial Sentiment Analysis","date":"2022-07-07","arxiv_id":"2207.03037","n_code_links":0,"syntology":null},{"paper":null,"slug":"ask-me-what-you-need-product-retrieval-using","title":"Ask Me What You Need: Product Retrieval using Knowledge from GPT-3","date":"2022-07-06","arxiv_id":"2207.02516","n_code_links":0,"syntology":null},{"paper":null,"slug":"aspect-based-sentiment-analysis-using-local","title":"Aspect-Based Sentiment Analysis using Local Context Focus Mechanism with DeBERTa","date":"2022-07-06","arxiv_id":"2207.02424","n_code_links":0,"syntology":null},{"paper":"/paper/simlm-pre-training-with-representation","slug":"simlm-pre-training-with-representation","title":"SimLM: Pre-training with Representation Bottleneck for Dense Passage Retrieval","date":"2022-07-06","arxiv_id":"2207.02578","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-role-of-complex-nlp-in-transformers-for","title":"The Role of Complex NLP in Transformers for Text Ranking?","date":"2022-07-06","arxiv_id":"2207.02522","n_code_links":0,"syntology":null},{"paper":"/paper/betti-numbers-of-attention-graphs-is-all-you-1","slug":"betti-numbers-of-attention-graphs-is-all-you-1","title":"Betti numbers of attention graphs is all you really need","date":"2022-07-05","arxiv_id":"2207.01903","n_code_links":1,"syntology":null},{"paper":null,"slug":"machine-learning-model-sizes-and-the","title":"Machine Learning Model Sizes and the Parameter Gap","date":"2022-07-05","arxiv_id":"2207.02852","n_code_links":0,"syntology":null},{"paper":null,"slug":"bert-can-he-predict-contrastive-focus","title":"BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model","date":"2022-07-04","arxiv_id":"2207.01718","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-contextual-sentence-analysis-models-to","title":"Using contextual sentence analysis models to recognize ESG concepts","date":"2022-07-04","arxiv_id":"2207.01402","n_code_links":0,"syntology":null},{"paper":null,"slug":"guim-general-user-and-item-embedding-with","title":"GUIM -- General User and Item Embedding with Mixture of Representation in E-commerce","date":"2022-07-02","arxiv_id":"2207.00750","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-polyphone-bert-for-polyphone-disambiguation","title":"A Polyphone BERT for Polyphone Disambiguation in Mandarin Chinese","date":"2022-07-01","arxiv_id":"2207.12089","n_code_links":0,"syntology":null},{"paper":"/paper/anoshift-a-distribution-shift-benchmark-for","slug":"anoshift-a-distribution-shift-benchmark-for","title":"AnoShift: A Distribution Shift Benchmark for Unsupervised Anomaly Detection","date":"2022-06-30","arxiv_id":"2206.15476","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["bit-ml/anoshift"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"compressing-pre-trained-transformers-via-low","title":"Compressing Pre-trained Transformers via Low-Bit NxM Sparsity for Natural Language Understanding","date":"2022-06-30","arxiv_id":"2206.15014","n_code_links":0,"syntology":null},{"paper":"/paper/deepspeed-inference-enabling-efficient","slug":"deepspeed-inference-enabling-efficient","title":"DeepSpeed Inference: Enabling Efficient Inference of Transformer Models at Unprecedented Scale","date":"2022-06-30","arxiv_id":"2207.00032","n_code_links":2,"syntology":null},{"paper":"/paper/gaitforemer-self-supervised-pre-training-of","slug":"gaitforemer-self-supervised-pre-training-of","title":"GaitForeMer: Self-Supervised Pre-Training of Transformers via Human Motion Forecasting for Few-Shot Gait Impairment Severity Estimation","date":"2022-06-30","arxiv_id":"2207.00106","n_code_links":1,"syntology":null},{"paper":null,"slug":"listbert-learning-to-rank-e-commerce-products","title":"ListBERT: Learning to Rank E-commerce products with Listwise BERT","date":"2022-06-30","arxiv_id":"2206.15198","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-topological-bert-transforming-attention","title":"The Topological BERT: Transforming Attention into Topology for Natural Language Processing","date":"2022-06-30","arxiv_id":"2206.15195","n_code_links":0,"syntology":null},{"paper":"/paper/two-stage-classifier-for-covid-19","slug":"two-stage-classifier-for-covid-19","title":"Two-Stage Classifier for COVID-19 Misinformation Detection Using BERT: a Study on Indonesian Tweets","date":"2022-06-30","arxiv_id":"2206.15359","n_code_links":2,"syntology":null},{"paper":"/paper/chinese-word-sense-embedding-with-sememewsd","slug":"chinese-word-sense-embedding-with-sememewsd","title":"Chinese Word Sense Embedding with SememeWSD and Synonym Set","date":"2022-06-29","arxiv_id":"2206.14388","n_code_links":2,"syntology":null},{"paper":null,"slug":"salo-an-efficient-spatial-accelerator","title":"SALO: An Efficient Spatial Accelerator Enabling Hybrid Sparse Attention Mechanisms for Long Sequences","date":"2022-06-29","arxiv_id":"2206.14550","n_code_links":0,"syntology":null},{"paper":null,"slug":"simple-and-effective-multi-sentence-tts-with","title":"Simple and Effective Multi-sentence TTS with Expressive and Coherent Prosody","date":"2022-06-29","arxiv_id":"2206.14643","n_code_links":0,"syntology":null},{"paper":"/paper/summarizing-videos-using-concentrated","slug":"summarizing-videos-using-concentrated","title":"Summarizing Videos using Concentrated Attention and Considering the Uniqueness and Diversity of the Video Frames","date":"2022-06-29","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"two-stage-covid19-classification-using-bert","title":"Two-Stage COVID19 Classification Using BERT Features","date":"2022-06-29","arxiv_id":"2206.14861","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-linguistic-feature-and-model","title":"Exploring linguistic feature and model combination for speech recognition based automatic AD detection","date":"2022-06-28","arxiv_id":"2206.13758","n_code_links":0,"syntology":null},{"paper":"/paper/materials-transformers-language-models-for","slug":"materials-transformers-language-models-for","title":"Materials Transformers Language Models for Generative Materials Design: a benchmark study","date":"2022-06-27","arxiv_id":"2206.13578","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-model-based-deep-learning-framework","title":"A multi-model-based deep learning framework for short text multiclass classification with the imbalanced and extremely small data set","date":"2022-06-24","arxiv_id":"2206.12027","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-test-for-evaluating-performance-in-human","title":"A Test for Evaluating Performance in Human-Computer Systems","date":"2022-06-24","arxiv_id":"2206.12390","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-and-author-level-political-inference","title":"Text and author-level political inference using heterogeneous knowledge representations","date":"2022-06-24","arxiv_id":"2206.12293","n_code_links":0,"syntology":null},{"paper":null,"slug":"unified-bert-for-few-shot-natural-language","title":"Unified BERT for Few-shot Natural Language Understanding","date":"2022-06-24","arxiv_id":"2206.12094","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-bert-embeddings-to-model-word-1","title":"Using BERT Embeddings to Model Word Importance in Conversational Transcripts for Deaf and Hard of Hearing Users","date":"2022-06-24","arxiv_id":"2206.12368","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-disability-lens-towards-biases-in-gpt-3","title":"A Disability Lens towards Biases in GPT-3 Generated Open-Ended Languages","date":"2022-06-23","arxiv_id":"2206.11993","n_code_links":0,"syntology":null},{"paper":"/paper/bert-rankers-are-brittle-a-study-using","slug":"bert-rankers-are-brittle-a-study-using","title":"BERT Rankers are Brittle: a Study using Adversarial Document Perturbations","date":"2022-06-23","arxiv_id":"2206.11724","n_code_links":1,"syntology":null},{"paper":"/paper/revisiting-orthogonality-regularization-a","slug":"revisiting-orthogonality-regularization-a","title":"Revisiting Orthogonality Regularization: A Study for Convolutional Neural Networks in Image Classification","date":"2022-06-23","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-winoqueer-developing-a-benchmark-for","title":"Towards WinoQueer: Developing a Benchmark for Anti-Queer Bias in Large Language Models","date":"2022-06-23","arxiv_id":"2206.11484","n_code_links":0,"syntology":null},{"paper":null,"slug":"answer-fast-accelerating-bert-on-the-tensor","title":"Answer Fast: Accelerating BERT on the Tensor Streaming Processor","date":"2022-06-22","arxiv_id":"2206.11062","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-and-effective-training-of-language","title":"Efficient and effective training of language and graph neural network models","date":"2022-06-22","arxiv_id":"2206.10781","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-automatic-and-efficient-bert-pruning-for","title":"An Automatic and Efficient BERT Pruning for Edge AI Systems","date":"2022-06-21","arxiv_id":"2206.10461","n_code_links":0,"syntology":null},{"paper":null,"slug":"cocopie-xgen-a-full-stack-ai-oriented","title":"CoCoPIE XGen: A Full-Stack AI-Oriented Optimizing Framework","date":"2022-06-21","arxiv_id":"2206.10620","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-fusion-for-language-model","title":"Knowledge Graph Fusion for Language Model Fine-tuning","date":"2022-06-21","arxiv_id":"2206.14574","n_code_links":0,"syntology":null},{"paper":null,"slug":"taphsir-towards-anaphoric-ambiguity-detection","title":"TAPHSIR: Towards AnaPHoric Ambiguity Detection and ReSolution In Requirements","date":"2022-06-21","arxiv_id":"2206.10227","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-cognitive-psychology-to-understand-gpt","title":"Using cognitive psychology to understand GPT-3","date":"2022-06-21","arxiv_id":"2206.14576","n_code_links":0,"syntology":null}],"record_sha256":"0752f0df266a9f13619b8a5d06b7fe294a489766389cbe22778859ad1adfdeff","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}