{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/bpe/papers/171","list_of":"/method/bpe","method":"BPE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":171,"pages_in_order":190,"rows_per_page":100,"rows":[17001,17100],"of":18975,"counts":{"archive_papers_tagged":18975,"with_a_code_link":8675,"where_syntology_ran_a_sample":2895,"not_listed_spam_title":0,"listed":18975,"listed_where_code_ran":2895,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2443,"every_run_a_failure_of_syntologys_instrument":452,"listed_with_a_run_with_no_instrument_failure":2443,"listed_every_run_a_failure_of_syntologys_instrument":452,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/bpe","prev":"/method/bpe/papers/170","next":"/method/bpe/papers/172","papers":[{"paper":null,"slug":"deep-deformation-detail-synthesis-for-thin","title":"Deep Deformation Detail Synthesis for Thin Shell Models","date":"2021-02-23","arxiv_id":"2102.11541","n_code_links":0,"syntology":null},{"paper":"/paper/do-transformer-modifications-transfer-across","slug":"do-transformer-modifications-transfer-across","title":"Do Transformer Modifications Transfer Across Implementations and Applications?","date":"2021-02-23","arxiv_id":"2102.11972","n_code_links":1,"syntology":null},{"paper":null,"slug":"robust-and-transferable-anomaly-detection-in","title":"Robust and Transferable Anomaly Detection in Log Data using Pre-Trained Language Models","date":"2021-02-23","arxiv_id":"2102.11570","n_code_links":0,"syntology":null},{"paper":"/paper/deepfake-video-detection-using-convolutional","slug":"deepfake-video-detection-using-convolutional","title":"Deepfake Video Detection Using Convolutional Vision Transformer","date":"2021-02-22","arxiv_id":"2102.11126","n_code_links":1,"syntology":null},{"paper":null,"slug":"determination-of-fault-location-in","title":"Determination of Fault Location in Transmission Lines with Image Processing and Artificial Neural Networks","date":"2021-02-22","arxiv_id":"2102.11073","n_code_links":0,"syntology":null},{"paper":"/paper/do-we-really-need-explicit-position-encodings","slug":"do-we-really-need-explicit-position-encodings","title":"Conditional Positional Encodings for Vision Transformers","date":"2021-02-22","arxiv_id":"2102.10882","n_code_links":2,"syntology":null},{"paper":null,"slug":"few-shot-learning-for-information","title":"Few Shot Learning for Information Verification","date":"2021-02-22","arxiv_id":"2102.10956","n_code_links":0,"syntology":null},{"paper":null,"slug":"position-information-in-transformers-an","title":"Position Information in Transformers: An Overview","date":"2021-02-22","arxiv_id":"2102.11090","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-is-all-you-need-multimodal","slug":"transformer-is-all-you-need-multimodal","title":"UniT: Multimodal Multitask Learning with a Unified Transformer","date":"2021-02-22","arxiv_id":"2102.10772","n_code_links":1,"syntology":null},{"paper":"/paper/medical-transformer-gated-axial-attention-for","slug":"medical-transformer-gated-axial-attention-for","title":"Medical Transformer: Gated Axial-Attention for Medical Image Segmentation","date":"2021-02-21","arxiv_id":"2102.10662","n_code_links":2,"syntology":null},{"paper":null,"slug":"multilingual-answer-sentence-reranking-via","title":"Multilingual Answer Sentence Reranking via Automatically Translated Data","date":"2021-02-20","arxiv_id":"2102.10250","n_code_links":0,"syntology":null},{"paper":"/paper/towards-accurate-and-compact-architectures","slug":"towards-accurate-and-compact-architectures","title":"Towards Accurate and Compact Architectures via Neural Architecture Transformer","date":"2021-02-20","arxiv_id":"2102.10301","n_code_links":2,"syntology":null},{"paper":"/paper/calibrate-before-use-improving-few-shot","slug":"calibrate-before-use-improving-few-shot","title":"Calibrate Before Use: Improving Few-Shot Performance of Language Models","date":"2021-02-19","arxiv_id":"2102.09690","n_code_links":5,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"0 ran · 4 unverified","official":{"repos":["tonyzhaozh/few-shot-learning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":null,"slug":"dialect-identification-in-nuanced-arabic","title":"Dialect Identification in Nuanced Arabic Tweets Using Farasa Segmentation and AraBERT","date":"2021-02-19","arxiv_id":"2102.09749","n_code_links":0,"syntology":null},{"paper":"/paper/latent-variable-nested-set-transformers","slug":"latent-variable-nested-set-transformers","title":"Latent Variable Sequential Set Transformers For Joint Multi-Agent Motion Prediction","date":"2021-02-19","arxiv_id":"2104.00563","n_code_links":2,"syntology":{"ran":11,"of":17,"n_ran_checked":10,"n_instrument":1,"unverified":6,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["roggirg/AutoBots"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/going-full-tilt-boogie-on-document","slug":"going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","arxiv_id":"2102.09550","n_code_links":1,"syntology":null},{"paper":"/paper/quiz-style-question-generation-for-news","slug":"quiz-style-question-generation-for-news","title":"Quiz-Style Question Generation for News Stories","date":"2021-02-18","arxiv_id":"2102.09094","n_code_links":2,"syntology":null},{"paper":"/paper/beyond-fully-connected-layers-with","slug":"beyond-fully-connected-layers-with","title":"Beyond Fully-Connected Layers with Quaternions: Parameterization of Hypercomplex Multiplications with $1/n$ Parameters","date":"2021-02-17","arxiv_id":"2102.08597","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["astonzhang/Parameterization-of-Hypercomplex-Multiplications"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"theaitre-1-0-interactive-generation-of","title":"THEaiTRE 1.0: Interactive generation of theatre play scripts","date":"2021-02-17","arxiv_id":"2102.08892","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-transformers-in-natural-language","slug":"exploring-transformers-in-natural-language","title":"Exploring Transformers in Natural Language Generation: GPT, BERT, and XLNet","date":"2021-02-16","arxiv_id":"2102.08036","n_code_links":1,"syntology":null},{"paper":"/paper/gradinit-learning-to-initialize-neural","slug":"gradinit-learning-to-initialize-neural","title":"GradInit: Learning to Initialize Neural Networks for Stable and Efficient Training","date":"2021-02-16","arxiv_id":"2102.08098","n_code_links":2,"syntology":{"ran":7,"of":13,"n_ran_checked":4,"n_instrument":3,"unverified":6,"pointer_only":12,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["zhuchen03/gradinit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/revisiting-language-encoding-in-learning","slug":"revisiting-language-encoding-in-learning","title":"Revisiting Language Encoding in Learning Multilingual Representations","date":"2021-02-16","arxiv_id":"2102.08357","n_code_links":1,"syntology":null},{"paper":"/paper/terapipe-token-level-pipeline-parallelism-for","slug":"terapipe-token-level-pipeline-parallelism-for","title":"TeraPipe: Token-Level Pipeline Parallelism for Training Large-Scale Language Models","date":"2021-02-16","arxiv_id":"2102.07988","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":3,"n_instrument":4,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zhuohan123/terapipe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prompt-programming-for-large-language-models","title":"Prompt Programming for Large Language Models: Beyond the Few-Shot Paradigm","date":"2021-02-15","arxiv_id":"2102.07350","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-corruptive-force-of-ai-generated-advice","title":"The corruptive force of AI-generated advice","date":"2021-02-15","arxiv_id":"2102.07536","n_code_links":0,"syntology":null},{"paper":"/paper/translational-equivariance-in-kernelizable","slug":"translational-equivariance-in-kernelizable","title":"Translational Equivariance in Kernelizable Attention","date":"2021-02-15","arxiv_id":"2102.07680","n_code_links":1,"syntology":null},{"paper":null,"slug":"dancing-along-battery-enabling-transformer","title":"Dancing along Battery: Enabling Transformer with Run-time Reconfigurability on Mobile Devices","date":"2021-02-12","arxiv_id":"2102.06336","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-zero-shot-neural-machine","title":"Improving Zero-shot Neural Machine Translation on Language-specific Encoders-Decoders","date":"2021-02-12","arxiv_id":"2102.06578","n_code_links":0,"syntology":null},{"paper":null,"slug":"multiversal-views-on-language-models","title":"Multiversal views on language models","date":"2021-02-12","arxiv_id":"2102.06391","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-inference-performance-of","title":"Optimizing Inference Performance of Transformers on CPUs","date":"2021-02-12","arxiv_id":"2102.06621","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-language-models-with-lstm-based","slug":"transformer-language-models-with-lstm-based","title":"Transformer Language Models with LSTM-based Cross-utterance Information Representation","date":"2021-02-12","arxiv_id":"2102.06474","n_code_links":1,"syntology":null},{"paper":"/paper/proof-artifact-co-training-for-theorem","slug":"proof-artifact-co-training-for-theorem","title":"Proof Artifact Co-training for Theorem Proving with Language Models","date":"2021-02-11","arxiv_id":"2102.06203","n_code_links":4,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jasonrute/lean-proof-recording-public","jasonrute/lean_proof_recording","jesse-michael-han/lean-step-public"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"text-compression-aided-transformer-encoding","title":"Text Compression-aided Transformer Encoding","date":"2021-02-11","arxiv_id":"2102.05951","n_code_links":0,"syntology":null},{"paper":"/paper/nast-non-autoregressive-spatial-temporal","slug":"nast-non-autoregressive-spatial-temporal","title":"NAST: Non-Autoregressive Spatial-Temporal Transformer for Time Series Forecasting","date":"2021-02-10","arxiv_id":"2102.05624","n_code_links":1,"syntology":null},{"paper":"/paper/augpt-dialogue-with-pre-trained-language","slug":"augpt-dialogue-with-pre-trained-language","title":"AuGPT: Auxiliary Tasks and Data Augmentation for End-To-End Dialogue with Pre-Trained Language Models","date":"2021-02-09","arxiv_id":"2102.05126","n_code_links":1,"syntology":null},{"paper":null,"slug":"bayesian-transformer-language-models-for","title":"Bayesian Transformer Language Models for Speech Recognition","date":"2021-02-09","arxiv_id":"2102.04754","n_code_links":0,"syntology":null},{"paper":null,"slug":"conversational-query-rewriting-with-self","title":"Conversational Query Rewriting with Self-supervised Learning","date":"2021-02-09","arxiv_id":"2102.04708","n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-intent-detection-and-slot-filling-with","title":"Joint Intent Detection and Slot Filling with Wheel-Graph Attention Networks","date":"2021-02-09","arxiv_id":"2102.04610","n_code_links":0,"syntology":null},{"paper":"/paper/point-cloud-transformers-applied-to-collider","slug":"point-cloud-transformers-applied-to-collider","title":"Point Cloud Transformers applied to Collider Physics","date":"2021-02-09","arxiv_id":"2102.05073","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-hybrid-task-oriented-dialog-system-with","title":"A Hybrid Task-Oriented Dialog System with Domain and Task Adaptive Pretraining","date":"2021-02-08","arxiv_id":"2102.04506","n_code_links":0,"syntology":null},{"paper":"/paper/colorization-transformer-1","slug":"colorization-transformer-1","title":"Colorization Transformer","date":"2021-02-08","arxiv_id":"2102.04432","n_code_links":2,"syntology":null},{"paper":null,"slug":"generating-fake-cyber-threat-intelligence","title":"Generating Fake Cyber Threat Intelligence Using Transformer-Based Models","date":"2021-02-08","arxiv_id":"2102.04351","n_code_links":0,"syntology":null},{"paper":"/paper/how-true-is-gpt-2-an-empirical-analysis-of","slug":"how-true-is-gpt-2-an-empirical-analysis-of","title":"Bias Out-of-the-Box: An Empirical Analysis of Intersectional Occupational Biases in Popular Generative Language Models","date":"2021-02-08","arxiv_id":"2102.04130","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["oxai/intersectional_gpt2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/transreid-transformer-based-object-re","slug":"transreid-transformer-based-object-re","title":"TransReID: Transformer-based Object Re-Identification","date":"2021-02-08","arxiv_id":"2102.04378","n_code_links":4,"syntology":null},{"paper":"/paper/transunet-transformers-make-strong-encoders","slug":"transunet-transformers-make-strong-encoders","title":"TransUNet: Transformers Make Strong Encoders for Medical Image Segmentation","date":"2021-02-08","arxiv_id":"2102.04306","n_code_links":22,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Beckschen/TransUNet"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"wake-word-detection-with-streaming","title":"Wake Word Detection with Streaming Transformers","date":"2021-02-08","arxiv_id":"2102.04488","n_code_links":0,"syntology":null},{"paper":"/paper/nystromformer-a-nystrom-based-algorithm-for","slug":"nystromformer-a-nystrom-based-algorithm-for","title":"Nyströmformer: A Nyström-Based Algorithm for Approximating Self-Attention","date":"2021-02-07","arxiv_id":"2102.03902","n_code_links":10,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mlpen/Nystromformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"jointly-improving-language-understanding-and","title":"Jointly Improving Language Understanding and Generation with Quality-Weighted Weak Supervision of Automatic Labeling","date":"2021-02-06","arxiv_id":"2102.03551","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-data-to-text-generation-with-lm-based","title":"Neural Data-to-Text Generation with LM-based Text Augmentation","date":"2021-02-06","arxiv_id":"2102.03556","n_code_links":0,"syntology":null},{"paper":"/paper/pipetransformer-automated-elastic-pipelining","slug":"pipetransformer-automated-elastic-pipelining","title":"PipeTransformer: Automated Elastic Pipelining for Distributed Training of Transformers","date":"2021-02-05","arxiv_id":"2102.03161","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Distributed-AI/PipeTransformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"understanding-emails-and-drafting-responses","title":"Understanding Emails and Drafting Responses -- An Approach Using GPT-3","date":"2021-02-05","arxiv_id":"2102.03062","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-capabilities-limitations","title":"Understanding the Capabilities, Limitations, and Societal Impact of Large Language Models","date":"2021-02-04","arxiv_id":"2102.02503","n_code_links":0,"syntology":null},{"paper":null,"slug":"mufasa-multimodal-fusion-architecture-search","title":"MUFASA: Multimodal Fusion Architecture Search for Electronic Health Records","date":"2021-02-03","arxiv_id":"2102.02340","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-transfer-learning-with-transformers","title":"Introduction to Neural Transfer Learning with Transformers for Social Science Text Analysis","date":"2021-02-03","arxiv_id":"2102.02111","n_code_links":0,"syntology":null},{"paper":"/paper/pitfalls-of-static-language-modelling","slug":"pitfalls-of-static-language-modelling","title":"Mind the Gap: Assessing Temporal Generalization in Neural Language Models","date":"2021-02-03","arxiv_id":"2102.01951","n_code_links":1,"syntology":null},{"paper":"/paper/relaxed-transformer-decoders-for-direct","slug":"relaxed-transformer-decoders-for-direct","title":"Relaxed Transformer Decoders for Direct Action Proposal Generation","date":"2021-02-03","arxiv_id":"2102.01894","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["MCG-NJU/RTD-Action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-natural-and-controllable-cross","title":"Towards Natural and Controllable Cross-Lingual Voice Conversion Based on Neural TTS Model and Phonetic Posteriorgram","date":"2021-02-03","arxiv_id":"2102.01991","n_code_links":0,"syntology":null},{"paper":"/paper/automated-query-reformulation-for-efficient","slug":"automated-query-reformulation-for-efficient","title":"Automated Query Reformulation for Efficient Search based on Query Logs From Stack Overflow","date":"2021-02-01","arxiv_id":"2102.00826","n_code_links":1,"syntology":null},{"paper":null,"slug":"gtae-graph-transformer-based-auto-encoders","title":"GTAE: Graph-Transformer based Auto-Encoders for Linguistic-Constrained Text Style Transfer","date":"2021-02-01","arxiv_id":"2102.00769","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-depression-related-to-cannabis-a-knowledge","title":"\"Is depression related to cannabis?\": A knowledge-infused model for Entity and Relation Extraction with Limited Supervision","date":"2021-02-01","arxiv_id":"2102.01222","n_code_links":0,"syntology":null},{"paper":"/paper/computational-performance-predictions-for","slug":"computational-performance-predictions-for","title":"A Runtime-Based Computational Performance Predictor for Deep Neural Network Training","date":"2021-01-31","arxiv_id":"2102.00527","n_code_links":1,"syntology":null},{"paper":"/paper/short-text-clustering-with-transformers","slug":"short-text-clustering-with-transformers","title":"Short Text Clustering with Transformers","date":"2021-01-31","arxiv_id":"2102.00541","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthesizing-monolingual-data-for-neural","title":"Synthesizing Monolingual Data for Neural Machine Translation","date":"2021-01-29","arxiv_id":"2101.12462","n_code_links":0,"syntology":null},{"paper":"/paper/transition-based-graph-decoder-for-neural","slug":"transition-based-graph-decoder-for-neural","title":"Enhancing the Transformer Decoder with Transition-based Syntax","date":"2021-01-29","arxiv_id":"2101.12640","n_code_links":1,"syntology":null},{"paper":null,"slug":"lstm-sakt-lstm-encoded-sakt-like-transformer","title":"LSTM-SAKT: LSTM-Encoded SAKT-like Transformer for Knowledge Tracing","date":"2021-01-28","arxiv_id":"2102.00845","n_code_links":0,"syntology":null},{"paper":"/paper/tokens-to-token-vit-training-vision","slug":"tokens-to-token-vit-training-vision","title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","date":"2021-01-28","arxiv_id":"2101.11986","n_code_links":13,"syntology":{"ran":21,"of":26,"n_ran_checked":21,"n_instrument":0,"unverified":5,"pointer_only":8,"phrase":"21 ran (of which 16 constructed an object rather than computing a result; 21 with no instrument failure: 1 honoured, 0 violated, 20 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yitu-opensource/T2T-ViT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"an-explainable-transformer-based-deep","title":"An explainable Transformer-based deep learning model for the prediction of incident heart failure","date":"2021-01-27","arxiv_id":"2101.11359","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-multi-task-multi-lingual-learning","slug":"exploring-multi-task-multi-lingual-learning","title":"Exploring multi-task multi-lingual learning of transformer models for hate speech and offensive speech identification in social media","date":"2021-01-27","arxiv_id":"2101.11155","n_code_links":1,"syntology":null},{"paper":null,"slug":"spatial-channel-transformer-network-for","title":"Spatial-Channel Transformer Network for Trajectory Prediction on the Traffic Scenes","date":"2021-01-27","arxiv_id":"2101.11472","n_code_links":0,"syntology":null},{"paper":null,"slug":"analyzing-zero-shot-cross-lingual-transfer-in","title":"Analyzing Zero-shot Cross-lingual Transfer in Supervised NLP Tasks","date":"2021-01-26","arxiv_id":"2101.10649","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-can-reflect-syntactic-structure-if","title":"Attention Can Reflect Syntactic Structure (If You Let It)","date":"2021-01-26","arxiv_id":"2101.10927","n_code_links":0,"syntology":null},{"paper":null,"slug":"cptr-full-transformer-network-for-image","title":"CPTR: Full Transformer Network for Image Captioning","date":"2021-01-26","arxiv_id":"2101.10804","n_code_links":0,"syntology":null},{"paper":null,"slug":"randomized-deep-structured-prediction-for","title":"Randomized Deep Structured Prediction for Discourse-Level Processing","date":"2021-01-25","arxiv_id":"2101.10435","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-time-series-forecasting-with","title":"Multi-Task Time Series Forecasting With Shared Attention","date":"2021-01-24","arxiv_id":"2101.09645","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-better-integration-of-fuzzy-matches","slug":"towards-a-better-integration-of-fuzzy-matches","title":"Towards a Better Integration of Fuzzy Matches in Neural Machine Translation through Data Augmentation","date":"2021-01-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/wangchanberta-pretraining-transformer-based","slug":"wangchanberta-pretraining-transformer-based","title":"WangchanBERTa: Pretraining transformer-based Thai Language Models","date":"2021-01-24","arxiv_id":"2101.09635","n_code_links":2,"syntology":null},{"paper":"/paper/training-multilingual-pre-trained-language","slug":"training-multilingual-pre-trained-language","title":"Training Multilingual Pre-trained Language Model with Byte-level Subwords","date":"2021-01-23","arxiv_id":"2101.09469","n_code_links":1,"syntology":null},{"paper":null,"slug":"bert-transformer-model-for-detecting-arabic","title":"BERT Transformer model for Detecting Arabic GPT2 Auto-Generated Tweets","date":"2021-01-22","arxiv_id":"2101.09345","n_code_links":0,"syntology":null},{"paper":null,"slug":"enriching-non-autoregressive-transformer-with","title":"Enriching Non-Autoregressive Transformer with Syntactic and SemanticStructures for Neural Machine Translation","date":"2021-01-22","arxiv_id":"2101.08942","n_code_links":0,"syntology":null},{"paper":"/paper/activity-graph-transformer-for-temporal","slug":"activity-graph-transformer-for-temporal","title":"Activity Graph Transformer for Temporal Action Localization","date":"2021-01-21","arxiv_id":"2101.08540","n_code_links":0,"syntology":null},{"paper":"/paper/daf-re-a-challenging-crowd-sourced-large","slug":"daf-re-a-challenging-crowd-sourced-large","title":"DAF:re: A Challenging, Crowd-Sourced, Large-Scale, Long-Tailed Dataset For Anime Character Recognition","date":"2021-01-21","arxiv_id":"2101.08674","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["arkel23/animesion"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/evaluating-multilingual-text-encoders-for","slug":"evaluating-multilingual-text-encoders-for","title":"Evaluating Multilingual Text Encoders for Unsupervised Cross-Lingual Retrieval","date":"2021-01-21","arxiv_id":"2101.08370","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rlitschk/EncoderCLIR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learn-to-dance-with-aist-music-conditioned-3d","slug":"learn-to-dance-with-aist-music-conditioned-3d","title":"AI Choreographer: Music Conditioned 3D Dance Generation with AIST++","date":"2021-01-21","arxiv_id":"2101.08779","n_code_links":1,"syntology":null},{"paper":null,"slug":"active-learning-for-sequence-tagging-with","title":"Active Learning for Sequence Tagging with Deep Pre-trained Models and Bayesian Uncertainty Estimates","date":"2021-01-20","arxiv_id":"2101.08133","n_code_links":0,"syntology":null},{"paper":"/paper/open-domain-conversational-search-assistant","slug":"open-domain-conversational-search-assistant","title":"Open-Domain Conversational Search Assistant with Transformers","date":"2021-01-20","arxiv_id":"2101.08197","n_code_links":1,"syntology":null},{"paper":"/paper/pgt-pseudo-relevance-feedback-using-a-graph","slug":"pgt-pseudo-relevance-feedback-using-a-graph","title":"PGT: Pseudo Relevance Feedback Using a Graph-Based Transformer","date":"2021-01-20","arxiv_id":"2101.07918","n_code_links":1,"syntology":null},{"paper":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","n_code_links":1,"syntology":null},{"paper":"/paper/fast-convergence-of-detr-with-spatially","slug":"fast-convergence-of-detr-with-spatially","title":"Fast Convergence of DETR with Spatially Modulated Co-Attention","date":"2021-01-19","arxiv_id":"2101.07448","n_code_links":2,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["gaopengcuhk/SMCA-DETR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/towards-facilitating-empathic-conversations","slug":"towards-facilitating-empathic-conversations","title":"Towards Facilitating Empathic Conversations in Online Mental Health Support: A Reinforcement Learning Approach","date":"2021-01-19","arxiv_id":"2101.07714","n_code_links":1,"syntology":null},{"paper":"/paper/inference-for-bart-with-multinomial-outcomes","slug":"inference-for-bart-with-multinomial-outcomes","title":"Inference for BART with Multinomial Outcomes","date":"2021-01-18","arxiv_id":"2101.06823","n_code_links":1,"syntology":null},{"paper":"/paper/dual-level-collaborative-transformer-for","slug":"dual-level-collaborative-transformer-for","title":"Dual-Level Collaborative Transformer for Image Captioning","date":"2021-01-16","arxiv_id":"2101.06462","n_code_links":1,"syntology":null},{"paper":"/paper/match-ignition-plugging-pagerank-into","slug":"match-ignition-plugging-pagerank-into","title":"Match-Ignition: Plugging PageRank into Transformer for Long-form Text Matching","date":"2021-01-16","arxiv_id":"2101.06423","n_code_links":1,"syntology":null},{"paper":null,"slug":"transformer-based-models-for-question","title":"Transformer-Based Models for Question Answering on COVID19","date":"2021-01-16","arxiv_id":"2101.11432","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploration-of-visual-features-and-their","title":"Exploration of Visual Features and their weighted-additive fusion for Video Captioning","date":"2021-01-14","arxiv_id":"2101.05806","n_code_links":0,"syntology":null},{"paper":"/paper/persistent-anti-muslim-bias-in-large-language","slug":"persistent-anti-muslim-bias-in-large-language","title":"Persistent Anti-Muslim Bias in Large Language Models","date":"2021-01-14","arxiv_id":"2101.05783","n_code_links":1,"syntology":null},{"paper":null,"slug":"privacy-analysis-in-language-models-via","title":"Training Data Leakage Analysis in Language Models","date":"2021-01-14","arxiv_id":"2101.05405","n_code_links":0,"syntology":null},{"paper":"/paper/coarse-and-fine-grained-hostility-detection","slug":"coarse-and-fine-grained-hostility-detection","title":"Coarse and Fine-Grained Hostility Detection in Hindi Posts using Fine Tuned Multilingual Embeddings","date":"2021-01-13","arxiv_id":"2101.04998","n_code_links":1,"syntology":null},{"paper":null,"slug":"fake-news-detection-system-using-xlnet-model","title":"Fake News Detection System using XLNet model with Topic Distributions: CONSTRAINT@AAAI2021 Shared Task","date":"2021-01-12","arxiv_id":"2101.11425","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-news-recommendation-with-negative","title":"Neural News Recommendation with Negative Feedback","date":"2021-01-12","arxiv_id":"2101.04328","n_code_links":0,"syntology":null},{"paper":"/paper/bert-gt-cross-sentence-n-ary-relation","slug":"bert-gt-cross-sentence-n-ary-relation","title":"BERT-GT: Cross-sentence n-ary relation extraction with BERT and Graph Transformer","date":"2021-01-11","arxiv_id":"2101.04158","n_code_links":0,"syntology":null}],"record_sha256":"b79e276e4f9e10083c721e53549a0d651da5a5ef4300d029df65dc9a482749f0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}