{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/adam/papers/221","list_of":"/method/adam","method":"Adam","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":221,"pages_in_order":244,"rows_per_page":100,"rows":[22001,22100],"of":24390,"counts":{"archive_papers_tagged":24390,"with_a_code_link":10944,"where_syntology_ran_a_sample":3424,"not_listed_spam_title":0,"listed":24390,"listed_where_code_ran":3424,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2899,"every_run_a_failure_of_syntologys_instrument":525,"listed_with_a_run_with_no_instrument_failure":2899,"listed_every_run_a_failure_of_syntologys_instrument":525,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/adam","prev":"/method/adam/papers/220","next":"/method/adam/papers/222","papers":[{"paper":"/paper/modeling-graph-structure-via-relative","slug":"modeling-graph-structure-via-relative","title":"Modeling Graph Structure via Relative Position for Text Generation from Knowledge Graphs","date":"2020-06-16","arxiv_id":"2006.09242","n_code_links":0,"syntology":null},{"paper":"/paper/perl-pivot-based-domain-adaptation-for-pre","slug":"perl-pivot-based-domain-adaptation-for-pre","title":"PERL: Pivot-based Domain Adaptation for Pre-trained Deep Contextualized Embedding Models","date":"2020-06-16","arxiv_id":"2006.09075","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalable-cross-lingual-pivots-to-model","title":"Scalable Cross Lingual Pivots to Model Pronoun Gender for Translation","date":"2020-06-16","arxiv_id":"2006.08881","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-sppd-system-for-schema-guided-dialogue","title":"The SPPD System for Schema Guided Dialogue State Tracking Challenge","date":"2020-06-16","arxiv_id":"2006.09035","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-online-evolving-framework-for-advancing","title":"An online evolving framework for advancing reinforcement-learning based automated vehicle control","date":"2020-06-15","arxiv_id":"2006.08092","n_code_links":0,"syntology":null},{"paper":null,"slug":"cooking-is-all-about-people-comment","title":"Cooking Is All About People: Comment Classification On Cookery Channels Using BERT and Classification Models (Malayalam-English Mix-Code)","date":"2020-06-15","arxiv_id":"2007.04249","n_code_links":0,"syntology":null},{"paper":null,"slug":"differentiable-neural-architecture","title":"Differentiable Neural Architecture Transformation for Reproducible Architecture Improvement","date":"2020-06-15","arxiv_id":"2006.08231","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-of-end-to-end-asr-for-openstt","title":"Exploration of End-to-End ASR for OpenSTT -- Russian Open Speech-to-Text Dataset","date":"2020-06-15","arxiv_id":"2006.08274","n_code_links":0,"syntology":null},{"paper":"/paper/finbert-a-pretrained-language-model-for","slug":"finbert-a-pretrained-language-model-for","title":"FinBERT: A Pretrained Language Model for Financial Communications","date":"2020-06-15","arxiv_id":"2006.08097","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yya518/FinBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-grained-human-evaluation-of-transformer","slug":"fine-grained-human-evaluation-of-transformer","title":"Fine-grained Human Evaluation of Transformer and Recurrent Approaches to Neural Machine Translation for English-to-Chinese","date":"2020-06-15","arxiv_id":"2006.08297","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-image-summarization-textual-summary","title":"Multi-Image Summarization: Textual Summary from a Set of Cohesive Images","date":"2020-06-15","arxiv_id":"2006.08686","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-multi-property-extraction-and-beyond","title":"On the Multi-Property Extraction and Beyond","date":"2020-06-15","arxiv_id":"2006.08281","n_code_links":0,"syntology":null},{"paper":"/paper/ffr-v1-1-fon-french-neural-machine","slug":"ffr-v1-1-fon-french-neural-machine","title":"FFR v1.1: Fon-French Neural Machine Translation","date":"2020-06-14","arxiv_id":"2006.09217","n_code_links":1,"syntology":null},{"paper":null,"slug":"finest-bert-and-crosloengual-bert-less-is","title":"FinEst BERT and CroSloEngual BERT: less is more in multilingual models","date":"2020-06-14","arxiv_id":"2006.07890","n_code_links":0,"syntology":null},{"paper":null,"slug":"guided-transformer-leveraging-multiple","title":"Guided Transformer: Leveraging Multiple External Sources for Representation Learning in Conversational Search","date":"2020-06-13","arxiv_id":"2006.07548","n_code_links":0,"syntology":null},{"paper":"/paper/modelling-high-level-mathematical-reasoning","slug":"modelling-high-level-mathematical-reasoning","title":"IsarStep: a Benchmark for High-level Mathematical Reasoning","date":"2020-06-13","arxiv_id":"2006.09265","n_code_links":2,"syntology":{"ran":2,"of":8,"n_ran_checked":2,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":null,"slug":"temporal-fusion-network-for-temporal-action","title":"Temporal Fusion Network for Temporal Action Localization:Submission to ActivityNet Challenge 2020 (Task E)","date":"2020-06-13","arxiv_id":"2006.07520","n_code_links":0,"syntology":null},{"paper":null,"slug":"transferring-monolingual-model-to-low","title":"Transferring Monolingual Model to Low-Resource Language: The Case of Tigrinya","date":"2020-06-13","arxiv_id":"2006.07698","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-natural-language-processing","title":"Comparing Natural Language Processing Techniques for Alzheimer's Dementia Prediction in Spontaneous Speech","date":"2020-06-12","arxiv_id":"2006.07358","n_code_links":0,"syntology":null},{"paper":null,"slug":"human-and-multi-agent-collaboration-in-a","title":"Human and Multi-Agent collaboration in a human-MARL teaming framework","date":"2020-06-12","arxiv_id":"2006.07301","n_code_links":0,"syntology":null},{"paper":"/paper/momentumrnn-integrating-momentum-into","slug":"momentumrnn-integrating-momentum-into","title":"MomentumRNN: Integrating Momentum into Recurrent Neural Networks","date":"2020-06-12","arxiv_id":"2006.06919","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["minhtannguyen/MomentumRNN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/unmasking-the-inductive-biases-of","slug":"unmasking-the-inductive-biases-of","title":"Benchmarking Unsupervised Object Representations for Video Sequences","date":"2020-06-12","arxiv_id":"2006.07034","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ecker-lab/object-centric-representation-benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-monolingual-approach-to-contextualized-word","slug":"a-monolingual-approach-to-contextualized-word","title":"A Monolingual Approach to Contextualized Word Embeddings for Mid-Resource Languages","date":"2020-06-11","arxiv_id":"2006.06202","n_code_links":0,"syntology":null},{"paper":"/paper/adaptive-gradient-methods-converge-faster","slug":"adaptive-gradient-methods-converge-faster","title":"Adaptive Gradient Methods Converge Faster with Over-Parameterization (but you should do a line-search)","date":"2020-06-11","arxiv_id":"2006.06835","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/dance-revolution-long-sequence-dance","slug":"dance-revolution-long-sequence-dance","title":"Dance Revolution: Long-Term Dance Generation with Music via Curriculum Learning","date":"2020-06-11","arxiv_id":"2006.06119","n_code_links":0,"syntology":null},{"paper":"/paper/fastpitch-parallel-text-to-speech-with-pitch","slug":"fastpitch-parallel-text-to-speech-with-pitch","title":"FastPitch: Parallel Text-to-speech with Pitch Prediction","date":"2020-06-11","arxiv_id":"2006.06873","n_code_links":6,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["NVIDIA/DeepLearningExamples"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper"]}}},{"paper":null,"slug":"implicit-kernel-attention","title":"Implicit Kernel Attention","date":"2020-06-11","arxiv_id":"2006.06147","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiplicative-noise-and-heavy-tails-in","title":"Multiplicative noise and heavy tails in stochastic optimization","date":"2020-06-11","arxiv_id":"2006.06293","n_code_links":0,"syntology":null},{"paper":"/paper/traffic-transformer-capturing-the-continuity","slug":"traffic-transformer-capturing-the-continuity","title":"Traffic transformer: Capturing the continuity and periodicity of time series for traffic forecasting","date":"2020-06-11","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/training-generative-adversarial-networks-with-2","slug":"training-generative-adversarial-networks-with-2","title":"Training Generative Adversarial Networks with Limited Data","date":"2020-06-11","arxiv_id":"2006.06676","n_code_links":28,"syntology":{"ran":21,"of":29,"n_ran_checked":20,"n_instrument":1,"unverified":8,"pointer_only":5,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 2 honoured, 1 violated, 17 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["NVlabs/stylegan2-ada"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"extrapolation-for-large-batch-training-in","title":"Extrapolation for Large-batch Training in Deep Learning","date":"2020-06-10","arxiv_id":"2006.05720","n_code_links":0,"syntology":null},{"paper":"/paper/mc-bert-efficient-language-pre-training-via-a","slug":"mc-bert-efficient-language-pre-training-via-a","title":"MC-BERT: Efficient Language Pre-Training via a Meta Controller","date":"2020-06-10","arxiv_id":"2006.05744","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["MC-BERT/MC-BERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/revisiting-few-sample-bert-fine-tuning","slug":"revisiting-few-sample-bert-fine-tuning","title":"Revisiting Few-sample BERT Fine-tuning","date":"2020-06-10","arxiv_id":"2006.05987","n_code_links":1,"syntology":null},{"paper":"/paper/few-shot-generative-conversational-query","slug":"few-shot-generative-conversational-query","title":"Few-Shot Generative Conversational Query Rewriting","date":"2020-06-09","arxiv_id":"2006.05009","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thunlp/ConversationQueryRewriter"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"graph-aware-transformer-is-attention-all","title":"Graph-Aware Transformer: Is Attention All Graphs Need?","date":"2020-06-09","arxiv_id":"2006.05213","n_code_links":0,"syntology":null},{"paper":null,"slug":"hausamt-v1-0-towards-english-hausa-neural","title":"HausaMT v1.0: Towards English-Hausa Neural Machine Translation","date":"2020-06-09","arxiv_id":"2006.05014","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-paraphrase-generation-using-pre","title":"Unsupervised Paraphrase Generation using Pre-trained Language Models","date":"2020-06-09","arxiv_id":"2006.05477","n_code_links":0,"syntology":null},{"paper":"/paper/big-gans-are-watching-you-towards","slug":"big-gans-are-watching-you-towards","title":"Object Segmentation Without Labels with Large-Scale Generative Models","date":"2020-06-08","arxiv_id":"2006.04988","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-disconnected-manifolds-a-no-gans","title":"Learning disconnected manifolds: a no GANs land","date":"2020-06-08","arxiv_id":"2006.04596","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-count-words-in-fluent-speech","slug":"learning-to-count-words-in-fluent-speech","title":"Learning to Count Words in Fluent Speech enables Online Speech Recognition","date":"2020-06-08","arxiv_id":"2006.04928","n_code_links":1,"syntology":null},{"paper":"/paper/linformer-self-attention-with-linear","slug":"linformer-self-attention-with-linear","title":"Linformer: Self-Attention with Linear Complexity","date":"2020-06-08","arxiv_id":"2006.04768","n_code_links":3,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/fairseq"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"modeling-discourse-structure-for-document","title":"Modeling Discourse Structure for Document-level Neural Machine Translation","date":"2020-06-08","arxiv_id":"2006.04721","n_code_links":0,"syntology":null},{"paper":"/paper/multispeech-multi-speaker-text-to-speech-with","slug":"multispeech-multi-speaker-text-to-speech-with","title":"MultiSpeech: Multi-Speaker Text to Speech with Transformer","date":"2020-06-08","arxiv_id":"2006.04664","n_code_links":1,"syntology":{"ran":1,"of":4,"n_ran_checked":1,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":null,"slug":"o-n-connections-are-expressive-enough","title":"$O(n)$ Connections are Expressive Enough: Universal Approximability of Sparse Transformers","date":"2020-06-08","arxiv_id":"2006.04862","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-stability-of-fine-tuning-bert","slug":"on-the-stability-of-fine-tuning-bert","title":"On the Stability of Fine-tuning BERT: Misconceptions, Explanations, and Strong Baselines","date":"2020-06-08","arxiv_id":"2006.04884","n_code_links":2,"syntology":{"ran":13,"of":20,"n_ran_checked":9,"n_instrument":4,"unverified":7,"pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["uds-lsv/bert-stable-fine-tuning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/wat-zei-je-detecting-out-of-distribution","slug":"wat-zei-je-detecting-out-of-distribution","title":"Wat zei je? Detecting Out-of-Distribution Translations with Variational Transformers","date":"2020-06-08","arxiv_id":"2006.08344","n_code_links":1,"syntology":null},{"paper":"/paper/bert-loses-patience-fast-and-robust-inference","slug":"bert-loses-patience-fast-and-robust-inference","title":"BERT Loses Patience: Fast and Robust Inference with Early Exit","date":"2020-06-07","arxiv_id":"2006.04152","n_code_links":1,"syntology":null},{"paper":"/paper/learning-texture-transformer-network-for-1","slug":"learning-texture-transformer-network-for-1","title":"Learning Texture Transformer Network for Image Super-Resolution","date":"2020-06-07","arxiv_id":"2006.04139","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["researchmm/TTSR"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"medical-concept-normalization-in-user","title":"Medical Concept Normalization in User Generated Texts by Learning Target Concept Embeddings","date":"2020-06-07","arxiv_id":"2006.04014","n_code_links":0,"syntology":null},{"paper":"/paper/pre-training-polish-transformer-based","slug":"pre-training-polish-transformer-based","title":"Pre-training Polish Transformer-based Language Models at Scale","date":"2020-06-07","arxiv_id":"2006.04229","n_code_links":1,"syntology":null},{"paper":null,"slug":"challenges-and-thrills-of-legal-arguments","title":"Challenges and Thrills of Legal Arguments","date":"2020-06-06","arxiv_id":"2006.03773","n_code_links":0,"syntology":null},{"paper":"/paper/accelerating-natural-language-understanding","slug":"accelerating-natural-language-understanding","title":"Accelerating Natural Language Understanding in Task-Oriented Dialog","date":"2020-06-05","arxiv_id":"2006.03701","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-overview-of-neural-network-compression","title":"An Overview of Neural Network Compression","date":"2020-06-05","arxiv_id":"2006.03669","n_code_links":0,"syntology":null},{"paper":"/paper/deberta-decoding-enhanced-bert-with","slug":"deberta-decoding-enhanced-bert-with","title":"DeBERTa: Decoding-enhanced BERT with Disentangled Attention","date":"2020-06-05","arxiv_id":"2006.03654","n_code_links":14,"syntology":{"ran":4,"of":13,"n_ran_checked":3,"n_instrument":1,"unverified":9,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["microsoft/DeBERTa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","named_in_paper","unlocated"]}}},{"paper":"/paper/funnel-transformer-filtering-out-sequential","slug":"funnel-transformer-filtering-out-sequential","title":"Funnel-Transformer: Filtering out Sequential Redundancy for Efficient Language Processing","date":"2020-06-05","arxiv_id":"2006.03236","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["laiguokun/Funnel-Transformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gmat-global-memory-augmentation-for","slug":"gmat-global-memory-augmentation-for","title":"GMAT: Global Memory Augmentation for Transformers","date":"2020-06-05","arxiv_id":"2006.03274","n_code_links":1,"syntology":null},{"paper":"/paper/masked-language-modeling-for-proteins-via","slug":"masked-language-modeling-for-proteins-via","title":"Masked Language Modeling for Proteins via Linearly Scalable Long-Context Transformers","date":"2020-06-05","arxiv_id":"2006.03555","n_code_links":1,"syntology":null},{"paper":null,"slug":"udpipe-at-evalatin-2020-contextualized-1","title":"UDPipe at EvaLatin 2020: Contextualized Embeddings and Treebank Embeddings","date":"2020-06-05","arxiv_id":"2006.03687","n_code_links":0,"syntology":null},{"paper":"/paper/visual-transformers-token-based-image","slug":"visual-transformers-token-based-image","title":"Visual Transformers: Token-based Image Representation and Processing for Computer Vision","date":"2020-06-05","arxiv_id":"2006.03677","n_code_links":8,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"end-to-end-speech-translation-with-knowledge-1","title":"End-to-End Speech-Translation with Knowledge Distillation: FBK@IWSLT2020","date":"2020-06-04","arxiv_id":"2006.02965","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-distributed-training-with-adaptive","title":"Scaling Distributed Training with Adaptive Summation","date":"2020-06-04","arxiv_id":"2006.02924","n_code_links":0,"syntology":null},{"paper":"/paper/the-sofc-exp-corpus-and-neural-approaches-to","slug":"the-sofc-exp-corpus-and-neural-approaches-to","title":"The SOFC-Exp Corpus and Neural Approaches to Information Extraction in the Materials Science Domain","date":"2020-06-04","arxiv_id":"2006.03039","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-text-summarization-of-covid-19","slug":"automatic-text-summarization-of-covid-19","title":"Automatic Text Summarization of COVID-19 Medical Research Articles using BERT and GPT-2","date":"2020-06-03","arxiv_id":"2006.01997","n_code_links":1,"syntology":null},{"paper":null,"slug":"cnn-denoisers-as-non-local-filters-the-neural","title":"The Neural Tangent Link Between CNN Denoisers and Non-Local Filters","date":"2020-06-03","arxiv_id":"2006.02379","n_code_links":0,"syntology":null},{"paper":"/paper/optimizing-neural-networks-via-koopman","slug":"optimizing-neural-networks-via-koopman","title":"Optimizing Neural Networks via Koopman Operator Theory","date":"2020-06-03","arxiv_id":"2006.02361","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-pairwise-probe-for-understanding-bert-fine","title":"A Pairwise Probe for Understanding BERT Fine-Tuning on Machine Reading Comprehension","date":"2020-06-02","arxiv_id":"2006.01346","n_code_links":0,"syntology":null},{"paper":"/paper/acceleration-of-descent-based-optimization","slug":"acceleration-of-descent-based-optimization","title":"Carathéodory Sampling for Stochastic Gradient Descent","date":"2020-06-02","arxiv_id":"2006.01819","n_code_links":1,"syntology":null},{"paper":"/paper/bert-based-multilingual-machine-comprehension","slug":"bert-based-multilingual-machine-comprehension","title":"BERT Based Multilingual Machine Comprehension in English and Hindi","date":"2020-06-02","arxiv_id":"2006.01432","n_code_links":2,"syntology":null},{"paper":"/paper/exploring-cross-sentence-contexts-for-named","slug":"exploring-cross-sentence-contexts-for-named","title":"Exploring Cross-sentence Contexts for Named Entity Recognition with BERT","date":"2020-06-02","arxiv_id":"2006.01563","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jouniluoma/bert-ner-cmv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-predictive-power-of-neural-language","slug":"on-the-predictive-power-of-neural-language","title":"On the Predictive Power of Neural Language Models for Human Real-Time Comprehension Behavior","date":"2020-06-02","arxiv_id":"2006.01912","n_code_links":1,"syntology":null},{"paper":null,"slug":"position-masking-for-language-models","title":"Position Masking for Language Models","date":"2020-06-02","arxiv_id":"2006.05676","n_code_links":0,"syntology":null},{"paper":"/paper/question-answering-on-scholarly-knowledge","slug":"question-answering-on-scholarly-knowledge","title":"Question Answering on Scholarly Knowledge Graphs","date":"2020-06-02","arxiv_id":"2006.01527","n_code_links":0,"syntology":null},{"paper":"/paper/subjective-question-answering-deciphering-the","slug":"subjective-question-answering-deciphering-the","title":"Subjective Question Answering: Deciphering the inner workings of Transformers in the realm of subjectivity","date":"2020-06-02","arxiv_id":"2006.08342","n_code_links":1,"syntology":null},{"paper":null,"slug":"wikibert-models-deep-transfer-learning-for","title":"WikiBERT models: deep transfer learning for many languages","date":"2020-06-02","arxiv_id":"2006.01538","n_code_links":0,"syntology":null},{"paper":"/paper/a-u-net-based-discriminator-for-generative-1","slug":"a-u-net-based-discriminator-for-generative-1","title":"A U-Net Based Discriminator for Generative Adversarial Networks","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/adahessian-an-adaptive-second-order-optimizer","slug":"adahessian-an-adaptive-second-order-optimizer","title":"ADAHESSIAN: An Adaptive Second Order Optimizer for Machine Learning","date":"2020-06-01","arxiv_id":"2006.00719","n_code_links":4,"syntology":{"ran":5,"of":11,"n_ran_checked":4,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["amirgholami/adahessian"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"an-effective-contextual-language-modeling","title":"An Effective Contextual Language Modeling Framework for Speech Summarization with Augmented Features","date":"2020-06-01","arxiv_id":"2006.01189","n_code_links":0,"syntology":null},{"paper":null,"slug":"approche-de-g-en-eration-de-r-eponse-a-base","title":"Approche de g\\'en\\'eration de r\\'eponse \\`a base de transformers (Transformer based approach for answer generation)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/bert-based-ensembles-for-modeling-disclosure","slug":"bert-based-ensembles-for-modeling-disclosure","title":"BERT-based Ensembles for Modeling Disclosure and Support in Conversational Social Media Text","date":"2020-06-01","arxiv_id":"2006.01222","n_code_links":0,"syntology":null},{"paper":"/paper/context-based-transformer-models-for-answer","slug":"context-based-transformer-models-for-answer","title":"Context-based Transformer Models for Answer Sentence Selection","date":"2020-06-01","arxiv_id":"2006.01285","n_code_links":1,"syntology":null},{"paper":null,"slug":"conversational-machine-comprehension-a","title":"Conversational Machine Comprehension: a Literature Review","date":"2020-06-01","arxiv_id":"2006.00671","n_code_links":0,"syntology":null},{"paper":"/paper/emergence-of-separable-manifolds-in-deep","slug":"emergence-of-separable-manifolds-in-deep","title":"Emergence of Separable Manifolds in Deep Language Representations","date":"2020-06-01","arxiv_id":"2006.01095","n_code_links":1,"syntology":null},{"paper":null,"slug":"etude-des-variations-s-emantiques-a-travers","title":"\\'Etude des variations s\\'emantiques \\`a travers plusieurs dimensions (Studying semantic variations through several dimensions )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-learning-of-part-specific","title":"Few-Shot Learning of Part-Specific Probability Space for 3D Shape Segmentation","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/image-search-with-text-feedback-by","slug":"image-search-with-text-feedback-by","title":"Image Search With Text Feedback by Visiolinguistic Attention Learning","date":"2020-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"introduction-d-informations-s-emantiques-dans","title":"Introduction d'informations s\\'emantiques dans un syst\\`eme de reconnaissance de la parole (Despite spectacular advances in recent years, the Automatic Speech Recognition (ASR) systems still make mistakes, especially in noisy environments)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"les-mod-eles-de-langue-contextuels-camembert","title":"Les mod\\`eles de langue contextuels Camembert pour le fran\\ccais : impact de la taille et de l'h\\'et\\'erog\\'en\\'eit\\'e des donn\\'ees d'entrainement (C AMEM BERT Contextual Language Models for French: Impact of Training Data Size and Heterogeneity )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-architecture-search-with-reinforce-and","title":"Hyperparameter optimization with REINFORCE and Transformers","date":"2020-06-01","arxiv_id":"2006.00939","n_code_links":0,"syntology":null},{"paper":"/paper/online-versus-offline-nmt-quality-an-in-depth","slug":"online-versus-offline-nmt-quality-an-in-depth","title":"Online Versus Offline NMT Quality: An In-depth Analysis on English-German and German-English","date":"2020-06-01","arxiv_id":"2006.00814","n_code_links":1,"syntology":null},{"paper":null,"slug":"qu-apporte-bert-a-l-analyse-syntaxique-en","title":"Qu'apporte BERT \\`a l'analyse syntaxique en constituants discontinus ? Une suite de tests pour \\'evaluer les pr\\'edictions de structures syntaxiques discontinues en anglais (What does BERT contribute to discontinuous constituency parsing ? A test suite to evaluate discontinuous constituency structure predictions in English)","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"r-e-entra-iner-ou-entra-iner-soi-m-eme-strat","title":"R\\'e-entra\\^\\iner ou entra\\^\\iner soi-m\\^eme ? Strat\\'egies de pr\\'e-entra\\^\\inement de BERT en domaine m\\'edical (Re-train or train from scratch ? Pre-training strategies for BERT in the medical domain )","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"rdcface-radial-distortion-correction-for-face","title":"RDCFace: Radial Distortion Correction for Face Recognition","date":"2020-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-sparse-view-backprojection-via","title":"Unsupervised Sparse-view Backprojection via Convolutional and Spatial Transformer Networks","date":"2020-06-01","arxiv_id":"2006.01658","n_code_links":0,"syntology":null},{"paper":null,"slug":"when-bert-forgets-how-to-pos-amnesic-probing","title":"Amnesic Probing: Behavioral Explanation with Amnesic Counterfactuals","date":"2020-06-01","arxiv_id":"2006.00995","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-accelerated-stochastic-gradient-method","title":"A New Accelerated Stochastic Gradient Method with Momentum","date":"2020-05-31","arxiv_id":"2006.00423","n_code_links":0,"syntology":null},{"paper":null,"slug":"bpgc-at-semeval-2020-task-11-propaganda","title":"BPGC at SemEval-2020 Task 11: Propaganda Detection in News Articles with Multi-Granularity Knowledge Sharing and Linguistic Features based Ensemble Learning","date":"2020-05-31","arxiv_id":"2006.00593","n_code_links":0,"syntology":null},{"paper":null,"slug":"cnrl-at-semeval-2020-task-5-modelling-causal","title":"CNRL at SemEval-2020 Task 5: Modelling Causal Reasoning in Language with Multi-Head Self-Attention Weights based Counterfactual Detection","date":"2020-05-31","arxiv_id":"2006.00609","n_code_links":0,"syntology":null},{"paper":null,"slug":"judge-me-by-my-size-noun-do-you-yodalib-a","title":"\"Judge me by my size (noun), do you?'' YodaLib: A Demographic-Aware Humor Generation Framework","date":"2020-05-31","arxiv_id":"2006.00578","n_code_links":0,"syntology":null},{"paper":null,"slug":"lrg-at-semeval-2020-task-7-assessing-the","title":"LRG at SemEval-2020 Task 7: Assessing the Ability of BERT and Derivative Models to Perform Short-Edits based Humor Grading","date":"2020-05-31","arxiv_id":"2006.00607","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-entity-linking-a-survey-of-models","title":"Neural Entity Linking: A Survey of Models Based on Deep Learning","date":"2020-05-31","arxiv_id":"2006.00575","n_code_links":0,"syntology":null}],"record_sha256":"17377b7f1b7b62b1cf9c0c886ba15f904ef06b20a023b9a92ace760f4b0a8eec","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}