{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/9","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":142,"rows_per_page":100,"rows":[801,900],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/8","next":"/task/language-modeling/papers/10","papers":[{"url":"/paper/generative-spoken-language-modeling-from-raw","slug":"generative-spoken-language-modeling-from-raw","title":"Generative Spoken Language Modeling from Raw Audio","date":"2021-02-01","arxiv_id":"2102.01192","repositories_listed":2,"syntology":null},{"url":"/paper/muppet-massive-multi-task-representations","slug":"muppet-massive-multi-task-representations","title":"Muppet: Massive Multi-task Representations with Pre-Finetuning","date":"2021-01-26","arxiv_id":"2101.11038","repositories_listed":2,"syntology":null},{"url":"/paper/cross-document-language-modeling","slug":"cross-document-language-modeling","title":"CDLM: Cross-Document Language Modeling","date":"2021-01-02","arxiv_id":"2101.00406","repositories_listed":2,"syntology":null},{"url":"/paper/ernie-doc-the-retrospective-long-document","slug":"ernie-doc-the-retrospective-long-document","title":"ERNIE-Doc: A Retrospective Long-Document Modeling Transformer","date":"2020-12-31","arxiv_id":"2012.15688","repositories_listed":2,"syntology":null},{"url":"/paper/intrinsic-dimensionality-explains-the","slug":"intrinsic-dimensionality-explains-the","title":"Intrinsic Dimensionality Explains the Effectiveness of Language Model Fine-Tuning","date":"2020-12-22","arxiv_id":"2012.13255","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intrinsic-dimensionality-explains-the#ran","syntology_url":"https://syntology.ai/paper/2012.13255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.13255"}},"official":{"repos":["rabeehk/compacter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fusing-context-into-knowledge-graph-for","slug":"fusing-context-into-knowledge-graph-for","title":"Fusing Context Into Knowledge Graph for Commonsense Question Answering","date":"2020-12-09","arxiv_id":"2012.04808","repositories_listed":2,"syntology":null},{"url":"/paper/pgl-at-textgraphs-2020-shared-task","slug":"pgl-at-textgraphs-2020-shared-task","title":"PGL at TextGraphs 2020 Shared Task: Explanation Regeneration using Language and Graph Learning Methods","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/structformer-joint-unsupervised-induction-of-1","slug":"structformer-joint-unsupervised-induction-of-1","title":"StructFormer: Joint Unsupervised Induction of Dependency and Constituency Structure from Masked Language Modeling","date":"2020-12-01","arxiv_id":"2012.00857","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-how-bert-learns-to-identify","slug":"understanding-how-bert-learns-to-identify","title":"An Investigation of Language Model Interpretability via Sentence Editing","date":"2020-11-28","arxiv_id":"2011.14039","repositories_listed":2,"syntology":null},{"url":"/paper/the-zero-resource-speech-benchmark-2021","slug":"the-zero-resource-speech-benchmark-2021","title":"The Zero Resource Speech Benchmark 2021: Metrics and baselines for unsupervised spoken language modeling","date":"2020-11-23","arxiv_id":"2011.11588","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/the-zero-resource-speech-benchmark-2021#ran","syntology_url":"https://syntology.ai/paper/2011.11588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.11588"}},"official":{"repos":["bootphon/zerospeech2021_baseline"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/data-informed-global-sparseness-in-attention","slug":"data-informed-global-sparseness-in-attention","title":"Data-Informed Global Sparseness in Attention Mechanisms for Deep Neural Networks","date":"2020-11-20","arxiv_id":"2012.02030","repositories_listed":2,"syntology":null},{"url":"/paper/character-level-representations-improve-drs","slug":"character-level-representations-improve-drs","title":"Character-level Representations Improve DRS-based Semantic Parsing Even in the Age of BERT","date":"2020-11-09","arxiv_id":"2011.04308","repositories_listed":2,"syntology":null},{"url":"/paper/understanding-pre-trained-bert-for-aspect","slug":"understanding-pre-trained-bert-for-aspect","title":"Understanding Pre-trained BERT for Aspect-based Sentiment Analysis","date":"2020-10-31","arxiv_id":"2011.00169","repositories_listed":2,"syntology":null},{"url":"/paper/automatically-identifying-words-that-can","slug":"automatically-identifying-words-that-can","title":"Automatically Identifying Words That Can Serve as Labels for Few-Shot Text Classification","date":"2020-10-26","arxiv_id":"2010.13641","repositories_listed":2,"syntology":null},{"url":"/paper/ernie-gram-pre-training-with-explicitly-n","slug":"ernie-gram-pre-training-with-explicitly-n","title":"ERNIE-Gram: Pre-Training with Explicitly N-Gram Masked Language Modeling for Natural Language Understanding","date":"2020-10-23","arxiv_id":"2010.12148","repositories_listed":2,"syntology":null},{"url":"/paper/tweeteval-unified-benchmark-and-comparative","slug":"tweeteval-unified-benchmark-and-comparative","title":"TweetEval: Unified Benchmark and Comparative Evaluation for Tweet Classification","date":"2020-10-23","arxiv_id":"2010.12421","repositories_listed":2,"syntology":null},{"url":"/paper/text-classification-using-label-names-only-a","slug":"text-classification-using-label-names-only-a","title":"Text Classification Using Label Names Only: A Language Model Self-Training Approach","date":"2020-10-14","arxiv_id":"2010.07245","repositories_listed":2,"syntology":null},{"url":"/paper/pagsusuri-ng-rnn-based-transfer-learning","slug":"pagsusuri-ng-rnn-based-transfer-learning","title":"Pagsusuri ng RNN-based Transfer Learning Technique sa Low-Resource Language","date":"2020-10-13","arxiv_id":"2010.06447","repositories_listed":2,"syntology":null},{"url":"/paper/genaug-data-augmentation-for-finetuning-text","slug":"genaug-data-augmentation-for-finetuning-text","title":"GenAug: Data Augmentation for Finetuning Text Generators","date":"2020-10-05","arxiv_id":"2010.01794","repositories_listed":2,"syntology":null},{"url":"/paper/semi-supervised-speech-language-joint-pre-1","slug":"semi-supervised-speech-language-joint-pre-1","title":"SPLAT: Speech-Language Joint Pre-Training for Spoken Language Understanding","date":"2020-10-05","arxiv_id":"2010.02295","repositories_listed":2,"syntology":null},{"url":"/paper/sentimental-liar-extended-corpus-and-deep","slug":"sentimental-liar-extended-corpus-and-deep","title":"Sentimental LIAR: Extended Corpus and Deep Learning Models for Fake Claim Classification","date":"2020-09-01","arxiv_id":"2009.01047","repositories_listed":2,"syntology":null},{"url":"/paper/intermediate-training-of-bert-for-product","slug":"intermediate-training-of-bert-for-product","title":"Intermediate Training of BERT for Product Matching","date":"2020-08-31","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/glancing-transformer-for-non-autoregressive","slug":"glancing-transformer-for-non-autoregressive","title":"Glancing Transformer for Non-Autoregressive Neural Machine Translation","date":"2020-08-18","arxiv_id":"2008.07905","repositories_listed":2,"syntology":null},{"url":"/paper/hybrid-ranking-network-for-text-to-sql","slug":"hybrid-ranking-network-for-text-to-sql","title":"Hybrid Ranking Network for Text-to-SQL","date":"2020-08-11","arxiv_id":"2008.04759","repositories_listed":2,"syntology":null},{"url":"/paper/delight-very-deep-and-light-weight","slug":"delight-very-deep-and-light-weight","title":"DeLighT: Deep and Light-weight Transformer","date":"2020-08-03","arxiv_id":"2008.00623","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/delight-very-deep-and-light-weight#ran","syntology_url":"https://syntology.ai/paper/2008.00623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.00623"}},"official":{"repos":["sacmehta/delight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/domain-specific-language-model-pretraining","slug":"domain-specific-language-model-pretraining","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","date":"2020-07-31","arxiv_id":"2007.15779","repositories_listed":2,"syntology":null},{"url":"/paper/mirostat-a-perplexity-controlled-neural-text","slug":"mirostat-a-perplexity-controlled-neural-text","title":"Mirostat: A Neural Text Decoding Algorithm that Directly Controls Perplexity","date":"2020-07-29","arxiv_id":"2007.14966","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mirostat-a-perplexity-controlled-neural-text#ran","syntology_url":"https://syntology.ai/paper/2007.14966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.14966"}},"official":{"repos":["basusourya/mirostat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-lottery-ticket-hypothesis-for-pre-trained","slug":"the-lottery-ticket-hypothesis-for-pre-trained","title":"The Lottery Ticket Hypothesis for Pre-trained BERT Networks","date":"2020-07-23","arxiv_id":"2007.12223","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-lottery-ticket-hypothesis-for-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2007.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.12223"}},"official":{"repos":["TAMU-VITA/BERT-Tickets","VITA-Group/BERT-Tickets"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-spoken-language-representations-with-1","slug":"learning-spoken-language-representations-with-1","title":"Learning Spoken Language Representations with Neural Lattice Language Modeling","date":"2020-07-06","arxiv_id":"2007.02629","repositories_listed":2,"syntology":null},{"url":"/paper/processing-south-asian-languages-written-in-1","slug":"processing-south-asian-languages-written-in-1","title":"Processing South Asian Languages Written in the Latin Script: the Dakshina Dataset","date":"2020-07-02","arxiv_id":"2007.01176","repositories_listed":2,"syntology":null},{"url":"/paper/pre-training-via-paraphrasing","slug":"pre-training-via-paraphrasing","title":"Pre-training via Paraphrasing","date":"2020-06-26","arxiv_id":"2006.15020","repositories_listed":2,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":3,"n_honours":3,"n_violates":2,"n_no_contract":7,"n_pointer_only":3,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 3 honoured, 2 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pre-training-via-paraphrasing#ran","syntology_url":"https://syntology.ai/paper/2006.15020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.15020"}},"official":null}},{"url":"/paper/finbert-a-pretrained-language-model-for","slug":"finbert-a-pretrained-language-model-for","title":"FinBERT: A Pretrained Language Model for Financial Communications","date":"2020-06-15","arxiv_id":"2006.08097","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finbert-a-pretrained-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2006.08097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08097"}},"official":{"repos":["yya518/FinBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/massive-choice-ample-tasks-machamp-a-toolkit","slug":"massive-choice-ample-tasks-machamp-a-toolkit","title":"Massive Choice, Ample Tasks (MaChAmp): A Toolkit for Multi-task Learning in NLP","date":"2020-05-29","arxiv_id":"2005.14672","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/massive-choice-ample-tasks-machamp-a-toolkit#ran","syntology_url":"https://syntology.ai/paper/2005.14672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14672"}},"official":{"repos":["machamp-nlp/machamp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-to-text-pre-training-for-data-to-text","slug":"text-to-text-pre-training-for-data-to-text","title":"Text-to-Text Pre-Training for Data-to-Text Tasks","date":"2020-05-21","arxiv_id":"2005.10433","repositories_listed":2,"syntology":null},{"url":"/paper/discrete-optimization-for-unsupervised","slug":"discrete-optimization-for-unsupervised","title":"Discrete Optimization for Unsupervised Sentence Summarization with Word-Level Extraction","date":"2020-05-04","arxiv_id":"2005.01791","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/discrete-optimization-for-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2005.01791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.01791"}},"official":{"repos":["raphael-sch/HC_Sentence_Summarization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/on-faithfulness-and-factuality-in-abstractive","slug":"on-faithfulness-and-factuality-in-abstractive","title":"On Faithfulness and Factuality in Abstractive Summarization","date":"2020-05-02","arxiv_id":"2005.00661","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-faithfulness-and-factuality-in-abstractive#ran","syntology_url":"https://syntology.ai/paper/2005.00661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00661"}},"official":{"repos":["google-research-datasets/xsum_hallucination_annotations"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/unifiedqa-crossing-format-boundaries-with-a","slug":"unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","arxiv_id":"2005.00700","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unifiedqa-crossing-format-boundaries-with-a#ran","syntology_url":"https://syntology.ai/paper/2005.00700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00700"}},"official":{"repos":["allenai/unifiedqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lite-transformer-with-long-short-range","slug":"lite-transformer-with-long-short-range","title":"Lite Transformer with Long-Short Range Attention","date":"2020-04-24","arxiv_id":"2004.11886","repositories_listed":2,"syntology":null},{"url":"/paper/palm-pre-training-an-autoencoding","slug":"palm-pre-training-an-autoencoding","title":"PALM: Pre-training an Autoencoding&Autoregressive Language Model for Context-conditioned Generation","date":"2020-04-14","arxiv_id":"2004.07159","repositories_listed":2,"syntology":null},{"url":"/paper/injecting-numerical-reasoning-skills-into","slug":"injecting-numerical-reasoning-skills-into","title":"Injecting Numerical Reasoning Skills into Language Models","date":"2020-04-09","arxiv_id":"2004.04487","repositories_listed":2,"syntology":null},{"url":"/paper/bae-bert-based-adversarial-examples-for-text","slug":"bae-bert-based-adversarial-examples-for-text","title":"BAE: BERT-based Adversarial Examples for Text Classification","date":"2020-04-04","arxiv_id":"2004.01970","repositories_listed":2,"syntology":null},{"url":"/paper/meta-fine-tuning-neural-language-models-for","slug":"meta-fine-tuning-neural-language-models-for","title":"Meta Fine-Tuning Neural Language Models for Multi-Domain Text Mining","date":"2020-03-29","arxiv_id":"2003.13003","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-content-based-sparse-attention-with-1","slug":"efficient-content-based-sparse-attention-with-1","title":"Efficient Content-Based Sparse Attention with Routing Transformers","date":"2020-03-12","arxiv_id":"2003.05997","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-content-based-sparse-attention-with-1#ran","syntology_url":"https://syntology.ai/paper/2003.05997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05997"}},"official":null}},{"url":"/paper/progen-language-modeling-for-protein","slug":"progen-language-modeling-for-protein","title":"ProGen: Language Modeling for Protein Generation","date":"2020-03-08","arxiv_id":"2004.03497","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/progen-language-modeling-for-protein#ran","syntology_url":"https://syntology.ai/paper/2004.03497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.03497"}},"official":null}},{"url":"/paper/cluecorpus2020-a-large-scale-chinese-corpus","slug":"cluecorpus2020-a-large-scale-chinese-corpus","title":"CLUECorpus2020: A Large-scale Chinese Corpus for Pre-training Language Model","date":"2020-03-03","arxiv_id":"2003.01355","repositories_listed":2,"syntology":null},{"url":"/paper/second-order-optimization-made-practical","slug":"second-order-optimization-made-practical","title":"Scalable Second Order Optimization for Deep Learning","date":"2020-02-20","arxiv_id":"2002.09018","repositories_listed":2,"syntology":null},{"url":"/paper/univilm-a-unified-video-and-language-pre","slug":"univilm-a-unified-video-and-language-pre","title":"UniVL: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation","date":"2020-02-15","arxiv_id":"2002.06353","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/univilm-a-unified-video-and-language-pre#ran","syntology_url":"https://syntology.ai/paper/2002.06353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.06353"}},"official":{"repos":["microsoft/UniVL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/parsing-as-pretraining","slug":"parsing-as-pretraining","title":"Parsing as Pretraining","date":"2020-02-05","arxiv_id":"2002.01685","repositories_listed":2,"syntology":{"n":30,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":14,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/parsing-as-pretraining#ran","syntology_url":"https://syntology.ai/paper/2002.01685","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.01685"}},"official":{"repos":["aghie/parsing-as-pretraining"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/scaling-laws-for-neural-language-models","slug":"scaling-laws-for-neural-language-models","title":"Scaling Laws for Neural Language Models","date":"2020-01-23","arxiv_id":"2001.08361","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-neural-language-models#ran","syntology_url":"https://syntology.ai/paper/2001.08361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.08361"}},"official":null}},{"url":"/paper/olmpics-on-what-language-model-pre-training","slug":"olmpics-on-what-language-model-pre-training","title":"oLMpics -- On what Language Model Pre-training Captures","date":"2019-12-31","arxiv_id":"1912.13283","repositories_listed":2,"syntology":null},{"url":"/paper/explicit-sparse-transformer-concentrated","slug":"explicit-sparse-transformer-concentrated","title":"Explicit Sparse Transformer: Concentrated Attention Through Explicit Selection","date":"2019-12-25","arxiv_id":"1912.11637","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explicit-sparse-transformer-concentrated#ran","syntology_url":"https://syntology.ai/paper/1912.11637","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.11637"}},"official":{"repos":["lancopku/Explicit-Sparse-Transformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/bertje-a-dutch-bert-model","slug":"bertje-a-dutch-bert-model","title":"BERTje: A Dutch BERT Model","date":"2019-12-19","arxiv_id":"1912.09582","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bertje-a-dutch-bert-model#ran","syntology_url":"https://syntology.ai/paper/1912.09582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.09582"}},"official":{"repos":["wietsedv/bertje"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-feasible-framework-for-arbitrary-shaped","slug":"a-feasible-framework-for-arbitrary-shaped","title":"A Feasible Framework for Arbitrary-Shaped Scene Text Recognition","date":"2019-12-10","arxiv_id":"1912.04561","repositories_listed":2,"syntology":null},{"url":"/paper/bp-transformer-modelling-long-range-context","slug":"bp-transformer-modelling-long-range-context","title":"BP-Transformer: Modelling Long-Range Context via Binary Partitioning","date":"2019-11-11","arxiv_id":"1911.04070","repositories_listed":2,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/bp-transformer-modelling-long-range-context#ran","syntology_url":"https://syntology.ai/paper/1911.04070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.04070"}},"official":{"repos":["yzh119/BPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/distilling-the-knowledge-of-bert-for-text-1","slug":"distilling-the-knowledge-of-bert-for-text-1","title":"Distilling Knowledge Learned in BERT for Text Generation","date":"2019-11-10","arxiv_id":"1911.03829","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/distilling-the-knowledge-of-bert-for-text-1#ran","syntology_url":"https://syntology.ai/paper/1911.03829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.03829"}},"official":{"repos":["ChenRocks/Distill-BERT-Textgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-transformer-models-by-reordering","slug":"improving-transformer-models-by-reordering","title":"Improving Transformer Models by Reordering their Sublayers","date":"2019-11-10","arxiv_id":"1911.03864","repositories_listed":2,"syntology":null},{"url":"/paper/memory-augmented-recurrent-neural-networks","slug":"memory-augmented-recurrent-neural-networks","title":"Memory-Augmented Recurrent Neural Networks Can Learn Generalized Dyck Languages","date":"2019-11-08","arxiv_id":"1911.03329","repositories_listed":2,"syntology":null},{"url":"/paper/open-domain-web-keyphrase-extraction-beyond-1","slug":"open-domain-web-keyphrase-extraction-beyond-1","title":"Open Domain Web Keyphrase Extraction Beyond Language Modeling","date":"2019-11-06","arxiv_id":"1911.02671","repositories_listed":2,"syntology":null},{"url":"/paper/neural-generation-for-czech-data-and","slug":"neural-generation-for-czech-data-and","title":"Neural Generation for Czech: Data and Baselines","date":"2019-10-11","arxiv_id":"1910.05298","repositories_listed":2,"syntology":null},{"url":"/paper/structured-pruning-of-large-language-models","slug":"structured-pruning-of-large-language-models","title":"Structured Pruning of Large Language Models","date":"2019-10-10","arxiv_id":"1910.04732","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/structured-pruning-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/1910.04732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04732"}},"official":{"repos":["asappresearch/flop"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-merits-of-universal-language-model-fine","slug":"the-merits-of-universal-language-model-fine","title":"The merits of Universal Language Model Fine-tuning for Small Datasets -- a case with Dutch book reviews","date":"2019-10-02","arxiv_id":"1910.00896","repositories_listed":2,"syntology":null},{"url":"/paper/structural-language-models-for-any-code","slug":"structural-language-models-for-any-code","title":"Structural Language Models of Code","date":"2019-09-30","arxiv_id":"1910.00577","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/structural-language-models-for-any-code#ran","syntology_url":"https://syntology.ai/paper/1910.00577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.00577"}},"official":{"repos":["tech-srl/slm-code-generation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/fake-news-detection-using-deep-learning","slug":"fake-news-detection-using-deep-learning","title":"Fake news detection using Deep Learning","date":"2019-09-29","arxiv_id":"1910.03496","repositories_listed":2,"syntology":null},{"url":"/paper/mixout-effective-regularization-to-finetune","slug":"mixout-effective-regularization-to-finetune","title":"Mixout: Effective Regularization to Finetune Large-scale Pretrained Language Models","date":"2019-09-25","arxiv_id":"1909.11299","repositories_listed":2,"syntology":null},{"url":"/paper/bertgrid-contextualized-embedding-for-2d","slug":"bertgrid-contextualized-embedding-for-2d","title":"BERTgrid: Contextualized Embedding for 2D Document Representation and Understanding","date":"2019-09-11","arxiv_id":"1909.04948","repositories_listed":2,"syntology":null},{"url":"/paper/taper-time-aware-patient-ehr-representation","slug":"taper-time-aware-patient-ehr-representation","title":"TAPER: Time-Aware Patient EHR Representation","date":"2019-08-11","arxiv_id":"1908.03971","repositories_listed":2,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/taper-time-aware-patient-ehr-representation#ran","syntology_url":"https://syntology.ai/paper/1908.03971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.03971"}},"official":{"repos":["sajaddarabi/TAPER","sajaddarabi/TAPER-EHR"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/what-bert-is-not-lessons-from-a-new-suite-of","slug":"what-bert-is-not-lessons-from-a-new-suite-of","title":"What BERT is not: Lessons from a new suite of psycholinguistic diagnostics for language models","date":"2019-07-31","arxiv_id":"1907.13528","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/what-bert-is-not-lessons-from-a-new-suite-of#ran","syntology_url":"https://syntology.ai/paper/1907.13528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.13528"}},"official":{"repos":["aetting/lm-diagnostics"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/caire-an-end-to-end-empathetic-chatbot","slug":"caire-an-end-to-end-empathetic-chatbot","title":"CAiRE: An Empathetic Neural Chatbot","date":"2019-07-28","arxiv_id":"1907.12108","repositories_listed":2,"syntology":null},{"url":"/paper/to-tune-or-not-to-tune-how-about-the-best-of","slug":"to-tune-or-not-to-tune-how-about-the-best-of","title":"To Tune or Not To Tune? How About the Best of Both Worlds?","date":"2019-07-09","arxiv_id":"1907.05338","repositories_listed":2,"syntology":null},{"url":"/paper/augmenting-self-attention-with-persistent","slug":"augmenting-self-attention-with-persistent","title":"Augmenting Self-attention with Persistent Memory","date":"2019-07-02","arxiv_id":"1907.01470","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":0,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/augmenting-self-attention-with-persistent#ran","syntology_url":"https://syntology.ai/paper/1907.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01470"}},"official":null}},{"url":"/paper/evaluating-language-model-finetuning","slug":"evaluating-language-model-finetuning","title":"Evaluating Language Model Finetuning Techniques for Low-resource Languages","date":"2019-06-30","arxiv_id":"1907.00409","repositories_listed":2,"syntology":null},{"url":"/paper/emotionx-ku-bert-max-based-contextual-emotion","slug":"emotionx-ku-bert-max-based-contextual-emotion","title":"EmotionX-KU: BERT-Max based Contextual Emotion Classifier","date":"2019-06-27","arxiv_id":"1906.11565","repositories_listed":2,"syntology":null},{"url":"/paper/multilingual-named-entity-recognition-using-1","slug":"multilingual-named-entity-recognition-using-1","title":"Multilingual Named Entity Recognition Using Pretrained Embeddings, Attention Mechanism and NCRF","date":"2019-06-21","arxiv_id":"1906.09978","repositories_listed":2,"syntology":null},{"url":"/paper/towards-robust-named-entity-recognition-for","slug":"towards-robust-named-entity-recognition-for","title":"Towards Robust Named Entity Recognition for Historic German","date":"2019-06-18","arxiv_id":"1906.07592","repositories_listed":2,"syntology":null},{"url":"/paper/generating-question-answer-hierarchies","slug":"generating-question-answer-hierarchies","title":"Generating Question-Answer Hierarchies","date":"2019-06-06","arxiv_id":"1906.02622","repositories_listed":2,"syntology":null},{"url":"/paper/the-unreasonable-effectiveness-of-transformer","slug":"the-unreasonable-effectiveness-of-transformer","title":"The Unreasonable Effectiveness of Transformer Language Models in Grammatical Error Correction","date":"2019-06-04","arxiv_id":"1906.01733","repositories_listed":2,"syntology":null},{"url":"/paper/190600363","slug":"190600363","title":"Does It Make Sense? And Why? A Pilot Study for Sense Making and Explanation","date":"2019-06-02","arxiv_id":"1906.00363","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/190600363#ran","syntology_url":"https://syntology.ai/paper/1906.00363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.00363"}},"official":{"repos":["wangcunxiang/Sen-Making-and-Explanation"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/discrete-flows-invertible-generative-models","slug":"discrete-flows-invertible-generative-models","title":"Discrete Flows: Invertible Generative Models of Discrete Data","date":"2019-05-24","arxiv_id":"1905.10347","repositories_listed":2,"syntology":null},{"url":"/paper/mu-forcing-training-variational-recurrent","slug":"mu-forcing-training-variational-recurrent","title":"mu-Forcing: Training Variational Recurrent Autoencoders for Text Generation","date":"2019-05-24","arxiv_id":"1905.10072","repositories_listed":2,"syntology":null},{"url":"/paper/sample-efficient-text-summarization-using-a","slug":"sample-efficient-text-summarization-using-a","title":"Sample Efficient Text Summarization Using a Single Pre-Trained Transformer","date":"2019-05-21","arxiv_id":"1905.08836","repositories_listed":2,"syntology":null},{"url":"/paper/190506596","slug":"190506596","title":"Joint Source-Target Self Attention with Locality Constraints","date":"2019-05-16","arxiv_id":"1905.06596","repositories_listed":2,"syntology":null},{"url":"/paper/a-surprisingly-robust-trick-for-winograd","slug":"a-surprisingly-robust-trick-for-winograd","title":"A Surprisingly Robust Trick for Winograd Schema Challenge","date":"2019-05-15","arxiv_id":"1905.06290","repositories_listed":2,"syntology":null},{"url":"/paper/rwth-asr-systems-for-librispeech-hybrid-vs","slug":"rwth-asr-systems-for-librispeech-hybrid-vs","title":"RWTH ASR Systems for LibriSpeech: Hybrid vs Attention -- w/o Data Augmentation","date":"2019-05-08","arxiv_id":"1905.03072","repositories_listed":2,"syntology":null},{"url":"/paper/adversarial-dropout-for-recurrent-neural","slug":"adversarial-dropout-for-recurrent-neural","title":"Adversarial Dropout for Recurrent Neural Networks","date":"2019-04-22","arxiv_id":"1904.09816","repositories_listed":2,"syntology":null},{"url":"/paper/few-shot-nlg-with-pre-trained-language-model","slug":"few-shot-nlg-with-pre-trained-language-model","title":"Few-Shot NLG with Pre-Trained Language Model","date":"2019-04-21","arxiv_id":"1904.09521","repositories_listed":2,"syntology":null},{"url":"/paper/190409324","slug":"190409324","title":"Mask-Predict: Parallel Decoding of Conditional Masked Language Models","date":"2019-04-19","arxiv_id":"1904.09324","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/190409324#ran","syntology_url":"https://syntology.ai/paper/1904.09324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09324"}},"official":{"repos":["facebookresearch/Mask-Predict"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pun-generation-with-surprise","slug":"pun-generation-with-surprise","title":"Pun Generation with Surprise","date":"2019-04-15","arxiv_id":"1904.06828","repositories_listed":2,"syntology":null},{"url":"/paper/rare-words-a-major-problem-for-contextualized","slug":"rare-words-a-major-problem-for-contextualized","title":"Rare Words: A Major Problem for Contextualized Embeddings And How to Fix it by Attentive Mimicking","date":"2019-04-14","arxiv_id":"1904.06707","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rare-words-a-major-problem-for-contextualized#ran","syntology_url":"https://syntology.ai/paper/1904.06707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06707"}},"official":{"repos":["timoschick/am-for-bert","timoschick/one-token-approximation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cyclical-annealing-schedule-a-simple-approach","slug":"cyclical-annealing-schedule-a-simple-approach","title":"Cyclical Annealing Schedule: A Simple Approach to Mitigating KL Vanishing","date":"2019-03-25","arxiv_id":"1903.10145","repositories_listed":2,"syntology":null},{"url":"/paper/improving-lemmatization-of-non-standard","slug":"improving-lemmatization-of-non-standard","title":"Improving Lemmatization of Non-Standard Languages with Joint Learning","date":"2019-03-16","arxiv_id":"1903.06939","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-user-modeling-with-long-and-short","slug":"adaptive-user-modeling-with-long-and-short","title":"Adaptive User Modeling with Long and Short-Term Preferences for Personalized Recommendation","date":"2019-01-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/knowledge-representation-learning-a","slug":"knowledge-representation-learning-a","title":"Knowledge Representation Learning: A Quantitative Review","date":"2018-12-28","arxiv_id":"1812.10901","repositories_listed":2,"syntology":null},{"url":"/paper/compositional-language-understanding-with","slug":"compositional-language-understanding-with","title":"Compositional Language Understanding with Text-based Relational Reasoning","date":"2018-11-07","arxiv_id":"1811.02959","repositories_listed":2,"syntology":null},{"url":"/paper/universal-language-model-fine-tuning-with","slug":"universal-language-model-fine-tuning-with","title":"Universal Language Model Fine-Tuning with Subword Tokenization for Polish","date":"2018-10-24","arxiv_id":"1810.10222","repositories_listed":2,"syntology":null},{"url":"/paper/continual-learning-of-recurrent-neural","slug":"continual-learning-of-recurrent-neural","title":"Continual Learning of Recurrent Neural Networks by Locally Aligning Distributed Representations","date":"2018-10-17","arxiv_id":"1810.07411","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-learning-of-recurrent-neural#ran","syntology_url":"https://syntology.ai/paper/1810.07411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.07411"}},"official":{"repos":["AnkurMali/ContinualPTNCN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/structured-content-preservation-for","slug":"structured-content-preservation-for","title":"Structured Content Preservation for Unsupervised Text Style Transfer","date":"2018-10-15","arxiv_id":"1810.06526","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-content-preservation-for#ran","syntology_url":"https://syntology.ai/paper/1810.06526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06526"}},"official":{"repos":["YouzhiTian/Structured-Content-Preservation-for-Unsupervised-Text-Style-Transfer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/frage-frequency-agnostic-word-representation","slug":"frage-frequency-agnostic-word-representation","title":"FRAGE: Frequency-Agnostic Word Representation","date":"2018-09-18","arxiv_id":"1809.06858","repositories_listed":2,"syntology":null},{"url":"/paper/pyramidal-recurrent-unit-for-language","slug":"pyramidal-recurrent-unit-for-language","title":"Pyramidal Recurrent Unit for Language Modeling","date":"2018-08-27","arxiv_id":"1808.09029","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pyramidal-recurrent-unit-for-language#ran","syntology_url":"https://syntology.ai/paper/1808.09029","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09029"}},"official":{"repos":["sacmehta/PRU"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarially-regularising-neural-nli-models","slug":"adversarially-regularising-neural-nli-models","title":"Adversarially Regularising Neural NLI Models to Integrate Logical Background Knowledge","date":"2018-08-26","arxiv_id":"1808.08609","repositories_listed":2,"syntology":null},{"url":"/paper/relational-recurrent-neural-networks","slug":"relational-recurrent-neural-networks","title":"Relational recurrent neural networks","date":"2018-06-05","arxiv_id":"1806.01822","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/relational-recurrent-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1806.01822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01822"}},"official":null}}],"record_sha256":"1f5b3c2717bbbac7dfc1a914b7d2e1ad4c278e111c1d35655adae0401a5f03c8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}