{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/55","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":55,"pages_in_order":142,"rows_per_page":100,"rows":[5401,5500],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling/papers/54","next":"/task/language-modeling/papers/56","papers":[{"url":"/paper/story-ending-prediction-by-transferable-bert","slug":"story-ending-prediction-by-transferable-bert","title":"Story Ending Prediction by Transferable BERT","date":"2019-05-17","arxiv_id":"1905.07504","repositories_listed":1,"syntology":null},{"url":"/paper/imho-fine-tuning-improves-claim-detection","slug":"imho-fine-tuning-improves-claim-detection","title":"IMHO Fine-Tuning Improves Claim Detection","date":"2019-05-16","arxiv_id":"1905.07000","repositories_listed":1,"syntology":null},{"url":"/paper/online-normalization-for-training-neural","slug":"online-normalization-for-training-neural","title":"Online Normalization for Training Neural Networks","date":"2019-05-15","arxiv_id":"1905.05894","repositories_listed":1,"syntology":{"n":16,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/online-normalization-for-training-neural#ran","syntology_url":"https://syntology.ai/paper/1905.05894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.05894"}},"official":{"repos":["cerebras/online-normalization"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/routing-networks-and-the-challenges-of","slug":"routing-networks-and-the-challenges-of","title":"Routing Networks and the Challenges of Modular and Compositional Computation","date":"2019-04-29","arxiv_id":"1904.12774","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/routing-networks-and-the-challenges-of#ran","syntology_url":"https://syntology.ai/paper/1904.12774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12774"}},"official":null}},{"url":"/paper/several-experiments-on-investigating","slug":"several-experiments-on-investigating","title":"Several Experiments on Investigating Pretraining and Knowledge-Enhanced Models for Natural Language Inference","date":"2019-04-27","arxiv_id":"1904.12104","repositories_listed":1,"syntology":null},{"url":"/paper/190501976","slug":"190501976","title":"TextKD-GAN: Text Generation using KnowledgeDistillation and Generative Adversarial Networks","date":"2019-04-23","arxiv_id":"1905.01976","repositories_listed":1,"syntology":null},{"url":"/paper/190409408","slug":"190409408","title":"Language Models with Transformers","date":"2019-04-20","arxiv_id":"1904.09408","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/190409408#ran","syntology_url":"https://syntology.ai/paper/1904.09408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09408"}},"official":{"repos":["cgraywang/gluon-nlp-1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/suggestion-mining-from-online-reviews-using","slug":"suggestion-mining-from-online-reviews-using","title":"Suggestion Mining from Online Reviews using ULMFiT","date":"2019-04-19","arxiv_id":"1904.09076","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-evaluation-of-transformer-language","slug":"dynamic-evaluation-of-transformer-language","title":"Dynamic Evaluation of Transformer Language Models","date":"2019-04-17","arxiv_id":"1904.08378","repositories_listed":1,"syntology":null},{"url":"/paper/iit-bhu-varanasi-at-msr-srst-2018-a-language-1","slug":"iit-bhu-varanasi-at-msr-srst-2018-a-language-1","title":"IIT (BHU) Varanasi at MSR-SRST 2018: A Language Model Based Approach for Natural Language Generation","date":"2019-04-12","arxiv_id":"1904.06234","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-model-for-joint-chinese-word","slug":"a-unified-model-for-joint-chinese-word","title":"A Graph-based Model for Joint Chinese Word Segmentation and Dependency Parsing","date":"2019-04-09","arxiv_id":"1904.04697","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-augmented-language-model-and-its","slug":"knowledge-augmented-language-model-and-its","title":"Knowledge-Augmented Language Model and its Application to Unsupervised Named-Entity Recognition","date":"2019-04-09","arxiv_id":"1904.04458","repositories_listed":1,"syntology":null},{"url":"/paper/seq3-differentiable-sequence-to-sequence-to","slug":"seq3-differentiable-sequence-to-sequence-to","title":"SEQ^3: Differentiable Sequence-to-Sequence-to-Sequence Autoencoder for Unsupervised Abstractive Sentence Compression","date":"2019-04-07","arxiv_id":"1904.03651","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-recurrent-neural-network","slug":"unsupervised-recurrent-neural-network","title":"Unsupervised Recurrent Neural Network Grammars","date":"2019-04-07","arxiv_id":"1904.03746","repositories_listed":1,"syntology":null},{"url":"/paper/alternative-weighting-schemes-for-elmo","slug":"alternative-weighting-schemes-for-elmo","title":"Alternative Weighting Schemes for ELMo Embeddings","date":"2019-04-05","arxiv_id":"1904.02954","repositories_listed":1,"syntology":null},{"url":"/paper/riemannian-normalizing-flow-on-variational","slug":"riemannian-normalizing-flow-on-variational","title":"Riemannian Normalizing Flow on Variational Wasserstein Autoencoder for Text Modeling","date":"2019-04-04","arxiv_id":"1904.02399","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/riemannian-normalizing-flow-on-variational#ran","syntology_url":"https://syntology.ai/paper/1904.02399","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.02399"}},"official":{"repos":["kingofspace0wzz/wae-rnf-lm"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-domain-adaptation-of","slug":"unsupervised-domain-adaptation-of","title":"Unsupervised Domain Adaptation of Contextualized Embeddings for Sequence Labeling","date":"2019-04-04","arxiv_id":"1904.02817","repositories_listed":1,"syntology":null},{"url":"/paper/sart-similarity-analogies-and-relatedness-for","slug":"sart-similarity-analogies-and-relatedness-for","title":"SART - Similarity, Analogies, and Relatedness for Tatar Language: New Benchmark Datasets for Word Embeddings Evaluation","date":"2019-03-31","arxiv_id":"1904.00365","repositories_listed":1,"syntology":null},{"url":"/paper/pre-trained-language-model-representations","slug":"pre-trained-language-model-representations","title":"Pre-trained Language Model Representations for Language Generation","date":"2019-03-22","arxiv_id":"1903.09722","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-sequence-to-sequence-models-for","slug":"evaluating-sequence-to-sequence-models-for","title":"Evaluating Sequence-to-Sequence Models for Handwritten Text Recognition","date":"2019-03-18","arxiv_id":"1903.07377","repositories_listed":1,"syntology":null},{"url":"/paper/the-emergence-of-number-and-syntax-units-in","slug":"the-emergence-of-number-and-syntax-units-in","title":"The emergence of number and syntax units in LSTM language models","date":"2019-03-18","arxiv_id":"1903.07435","repositories_listed":1,"syntology":null},{"url":"/paper/maybe-deep-neural-networks-are-the-best","slug":"maybe-deep-neural-networks-are-the-best","title":"Maybe Deep Neural Networks are the Best Choice for Modeling Source Code","date":"2019-03-13","arxiv_id":"1903.05734","repositories_listed":1,"syntology":null},{"url":"/paper/partially-shuffling-the-training-data-to-1","slug":"partially-shuffling-the-training-data-to-1","title":"Partially Shuffling the Training Data to Improve Language Models","date":"2019-03-11","arxiv_id":"1903.04167","repositories_listed":1,"syntology":null},{"url":"/paper/props-probabilistic-personalization-of-black","slug":"props-probabilistic-personalization-of-black","title":"PROPS: Probabilistic personalization of black-box sequence models","date":"2019-03-05","arxiv_id":"1903.02013","repositories_listed":1,"syntology":null},{"url":"/paper/russian-language-datasets-in-the-digitial","slug":"russian-language-datasets-in-the-digitial","title":"Russian Language Datasets in the Digitial Humanities Domain and Their Evaluation with Word Embeddings","date":"2019-03-04","arxiv_id":"1903.08739","repositories_listed":1,"syntology":null},{"url":"/paper/alternating-synthetic-and-real-gradients-for","slug":"alternating-synthetic-and-real-gradients-for","title":"Alternating Synthetic and Real Gradients for Neural Language Modeling","date":"2019-02-27","arxiv_id":"1902.10630","repositories_listed":1,"syntology":null},{"url":"/paper/an-embarrassingly-simple-approach-for","slug":"an-embarrassingly-simple-approach-for","title":"An Embarrassingly Simple Approach for Transfer Learning from Pretrained Language Models","date":"2019-02-27","arxiv_id":"1902.10547","repositories_listed":1,"syntology":null},{"url":"/paper/zoho-at-semeval-2019-task-9-semi-supervised","slug":"zoho-at-semeval-2019-task-9-semi-supervised","title":"Zoho at SemEval-2019 Task 9: Semi-supervised Domain Adaptation using Tri-training for Suggestion Mining","date":"2019-02-27","arxiv_id":"1902.10623","repositories_listed":1,"syntology":null},{"url":"/paper/polyglot-contextual-representations-improve","slug":"polyglot-contextual-representations-improve","title":"Polyglot Contextual Representations Improve Crosslingual Transfer","date":"2019-02-26","arxiv_id":"1902.09697","repositories_listed":1,"syntology":null},{"url":"/paper/a-fully-differentiable-beam-search-decoder","slug":"a-fully-differentiable-beam-search-decoder","title":"A Fully Differentiable Beam Search Decoder","date":"2019-02-16","arxiv_id":"1902.06022","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/a-fully-differentiable-beam-search-decoder#ran","syntology_url":"https://syntology.ai/paper/1902.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.06022"}},"official":null}},{"url":"/paper/the-referential-reader-a-recurrent-entity","slug":"the-referential-reader-a-recurrent-entity","title":"The Referential Reader: A Recurrent Entity Network for Anaphora Resolution","date":"2019-02-05","arxiv_id":"1902.01541","repositories_listed":1,"syntology":null},{"url":"/paper/review-conversational-reading-comprehension","slug":"review-conversational-reading-comprehension","title":"Review Conversational Reading Comprehension","date":"2019-02-03","arxiv_id":"1902.00821","repositories_listed":1,"syntology":null},{"url":"/paper/a-generalized-language-model-in-tensor-space","slug":"a-generalized-language-model-in-tensor-space","title":"A Generalized Language Model in Tensor Space","date":"2019-01-31","arxiv_id":"1901.11167","repositories_listed":1,"syntology":null},{"url":"/paper/latent-normalizing-flows-for-discrete","slug":"latent-normalizing-flows-for-discrete","title":"Latent Normalizing Flows for Discrete Sequences","date":"2019-01-29","arxiv_id":"1901.10548","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/latent-normalizing-flows-for-discrete#ran","syntology_url":"https://syntology.ai/paper/1901.10548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.10548"}},"official":{"repos":["harvardnlp/TextFlow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/team-papelo-transformer-networks-at-fever","slug":"team-papelo-transformer-networks-at-fever","title":"Team Papelo: Transformer Networks at FEVER","date":"2019-01-08","arxiv_id":"1901.02534","repositories_listed":1,"syntology":null},{"url":"/paper/transfer-learning-from-language-models-to","slug":"transfer-learning-from-language-models-to","title":"Transfer learning from language models to image caption generators: Better models may not transfer better","date":"2019-01-01","arxiv_id":"1901.01216","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transfer-learning-from-language-models-to#ran","syntology_url":"https://syntology.ai/paper/1901.01216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01216"}},"official":{"repos":["mtanti/mtanti-phd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-weight-symmetry-in-deep-neural","slug":"exploring-weight-symmetry-in-deep-neural","title":"Exploring Weight Symmetry in Deep Neural Networks","date":"2018-12-28","arxiv_id":"1812.11027","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/exploring-weight-symmetry-in-deep-neural#ran","syntology_url":"https://syntology.ai/paper/1812.11027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.11027"}},"official":{"repos":["hushell/deep-symmetry"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/writer-aware-cnn-for-parsimonious-hmm-based","slug":"writer-aware-cnn-for-parsimonious-hmm-based","title":"Writer-Aware CNN for Parsimonious HMM-Based Offline Handwritten Chinese Text Recognition","date":"2018-12-24","arxiv_id":"1812.09809","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-one-grain-of-sand-in-the-desert","slug":"what-is-one-grain-of-sand-in-the-desert","title":"What Is One Grain of Sand in the Desert? Analyzing Individual Neurons in Deep NLP Models","date":"2018-12-21","arxiv_id":"1812.09355","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-neural-networks-with-pre-trained","slug":"recurrent-neural-networks-with-pre-trained","title":"Recurrent Neural Networks with Pre-trained Language Model Embedding for Slot Filling Task","date":"2018-12-12","arxiv_id":"1812.05199","repositories_listed":1,"syntology":null},{"url":"/paper/practical-text-classification-with-large-pre","slug":"practical-text-classification-with-large-pre","title":"Practical Text Classification With Large Pre-Trained Language Models","date":"2018-12-04","arxiv_id":"1812.01207","repositories_listed":1,"syntology":null},{"url":"/paper/multi-level-multimodal-common-semantic-space","slug":"multi-level-multimodal-common-semantic-space","title":"Multi-level Multimodal Common Semantic Space for Image-Phrase Grounding","date":"2018-11-28","arxiv_id":"1811.11683","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-transfer-learning-for-spoken","slug":"unsupervised-transfer-learning-for-spoken","title":"Unsupervised Transfer Learning for Spoken Language Understanding in Intelligent Agents","date":"2018-11-13","arxiv_id":"1811.05370","repositories_listed":1,"syntology":null},{"url":"/paper/effective-subtree-encoding-for-easy-first","slug":"effective-subtree-encoding-for-easy-first","title":"Effective Representation for Easy-First Dependency Parsing","date":"2018-11-08","arxiv_id":"1811.03511","repositories_listed":1,"syntology":null},{"url":"/paper/do-rnns-learn-human-like-abstract-word-order","slug":"do-rnns-learn-human-like-abstract-word-order","title":"Do RNNs learn human-like abstract word order preferences?","date":"2018-11-05","arxiv_id":"1811.01866","repositories_listed":1,"syntology":null},{"url":"/paper/mesh-tensorflow-deep-learning-for","slug":"mesh-tensorflow-deep-learning-for","title":"Mesh-TensorFlow: Deep Learning for Supercomputers","date":"2018-11-05","arxiv_id":"1811.02084","repositories_listed":1,"syntology":null},{"url":"/paper/sentence-encoders-on-stilts-supplementary","slug":"sentence-encoders-on-stilts-supplementary","title":"Sentence Encoders on STILTs: Supplementary Training on Intermediate Labeled-data Tasks","date":"2018-11-02","arxiv_id":"1811.01088","repositories_listed":1,"syntology":null},{"url":"/paper/juman-a-morphological-analysis-toolkit-for","slug":"juman-a-morphological-analysis-toolkit-for","title":"Juman++: A Morphological Analysis Toolkit for Scriptio Continua","date":"2018-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-the-end-to-end-solution-to-mandarin","slug":"on-the-end-to-end-solution-to-mandarin","title":"On the End-to-End Solution to Mandarin-English Code-switching Speech Recognition","date":"2018-11-01","arxiv_id":"1811.00241","repositories_listed":1,"syntology":null},{"url":"/paper/improving-machine-reading-comprehension-with","slug":"improving-machine-reading-comprehension-with","title":"Improving Machine Reading Comprehension with General Reading Strategies","date":"2018-10-31","arxiv_id":"1810.13441","repositories_listed":1,"syntology":null},{"url":"/paper/language-modeling-with-sparse-product-of","slug":"language-modeling-with-sparse-product-of","title":"Language Modeling with Sparse Product of Sememe Experts","date":"2018-10-29","arxiv_id":"1810.12387","repositories_listed":1,"syntology":null},{"url":"/paper/language-modeling-for-code-switching","slug":"language-modeling-for-code-switching","title":"Language Modeling for Code-Switching: Evaluation, Integration of Monolingual Data, and Discriminative Training","date":"2018-10-28","arxiv_id":"1810.11895","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-stochastic-gradient-descent-for","slug":"evolutionary-stochastic-gradient-descent-for","title":"Evolutionary Stochastic Gradient Descent for Optimization of Deep Neural Networks","date":"2018-10-16","arxiv_id":"1810.06773","repositories_listed":1,"syntology":null},{"url":"/paper/trellis-networks-for-sequence-modeling","slug":"trellis-networks-for-sequence-modeling","title":"Trellis Networks for Sequence Modeling","date":"2018-10-15","arxiv_id":"1810.06682","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trellis-networks-for-sequence-modeling#ran","syntology_url":"https://syntology.ai/paper/1810.06682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06682"}},"official":{"repos":["locuslab/trellisnet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-neural-word-segmentation-for","slug":"unsupervised-neural-word-segmentation-for","title":"Unsupervised Neural Word Segmentation for Chinese via Segmental Language Modeling","date":"2018-10-07","arxiv_id":"1810.03167","repositories_listed":1,"syntology":null},{"url":"/paper/learning-compressed-transforms-with-low","slug":"learning-compressed-transforms-with-low","title":"Learning Compressed Transforms with Low Displacement Rank","date":"2018-10-04","arxiv_id":"1810.02309","repositories_listed":1,"syntology":{"n":21,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/learning-compressed-transforms-with-low#ran","syntology_url":"https://syntology.ai/paper/1810.02309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02309"}},"official":{"repos":["HazyResearch/structured-nets"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hybrid-approach-to-automatic-corpus","slug":"a-hybrid-approach-to-automatic-corpus","title":"A Hybrid Approach to Automatic Corpus Generation for Chinese Spelling Check","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/diversity-promoting-gan-a-cross-entropy-based","slug":"diversity-promoting-gan-a-cross-entropy-based","title":"Diversity-Promoting GAN: A Cross-Entropy Based Generative Adversarial Network for Diversified Text Generation","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/prompsits-submission-to-wmt-2018-parallel","slug":"prompsits-submission-to-wmt-2018-parallel","title":"Prompsit's submission to WMT 2018 Parallel Corpus Filtering shared task","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/put-it-back-entity-typing-with-language-model","slug":"put-it-back-entity-typing-with-language-model","title":"Put It Back: Entity Typing with Language Model Enhancement","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-data-made-to-order-the-case-of","slug":"synthetic-data-made-to-order-the-case-of","title":"Synthetic Data Made to Order: The Case of Parsing","date":"2018-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-recurrent-binaryternary-weights","slug":"learning-recurrent-binaryternary-weights","title":"Learning Recurrent Binary/Ternary Weights","date":"2018-09-28","arxiv_id":"1809.11086","repositories_listed":1,"syntology":null},{"url":"/paper/controllable-neural-story-plot-generation-via","slug":"controllable-neural-story-plot-generation-via","title":"Controllable Neural Story Plot Generation via Reward Shaping","date":"2018-09-27","arxiv_id":"1809.10736","repositories_listed":1,"syntology":null},{"url":"/paper/document-informed-neural-autoregressive-topic","slug":"document-informed-neural-autoregressive-topic","title":"Document Informed Neural Autoregressive Topic Models with Distributional Prior","date":"2018-09-15","arxiv_id":"1809.06709","repositories_listed":1,"syntology":null},{"url":"/paper/rnns-as-psycholinguistic-subjects-syntactic","slug":"rnns-as-psycholinguistic-subjects-syntactic","title":"RNNs as psycholinguistic subjects: Syntactic state and grammatical dependency","date":"2018-09-05","arxiv_id":"1809.01329","repositories_listed":1,"syntology":null},{"url":"/paper/simple-fusion-return-of-the-language-model","slug":"simple-fusion-return-of-the-language-model","title":"Simple Fusion: Return of the Language Model","date":"2018-09-01","arxiv_id":"1809.00125","repositories_listed":1,"syntology":{"n":10,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/simple-fusion-return-of-the-language-model#ran","syntology_url":"https://syntology.ai/paper/1809.00125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00125"}},"official":null}},{"url":"/paper/spherical-latent-spaces-for-stable","slug":"spherical-latent-spaces-for-stable","title":"Spherical Latent Spaces for Stable Variational Autoencoders","date":"2018-08-31","arxiv_id":"1808.10805","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spherical-latent-spaces-for-stable#ran","syntology_url":"https://syntology.ai/paper/1808.10805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.10805"}},"official":{"repos":["jiacheng-xu/vmf_vae_nlp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-output-connection-for-a-high-rank","slug":"direct-output-connection-for-a-high-rank","title":"Direct Output Connection for a High-Rank Language Model","date":"2018-08-30","arxiv_id":"1808.10143","repositories_listed":1,"syntology":null},{"url":"/paper/a-neural-model-of-adaptation-in-reading","slug":"a-neural-model-of-adaptation-in-reading","title":"A Neural Model of Adaptation in Reading","date":"2018-08-29","arxiv_id":"1808.09930","repositories_listed":1,"syntology":null},{"url":"/paper/grammar-induction-with-neural-language-models","slug":"grammar-induction-with-neural-language-models","title":"Grammar Induction with Neural Language Models: An Unusual Replication","date":"2018-08-29","arxiv_id":"1808.10000","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantum-many-body-wave-function-inspired","slug":"a-quantum-many-body-wave-function-inspired","title":"A Quantum Many-body Wave Function Inspired Language Modeling Approach","date":"2018-08-28","arxiv_id":"1808.09891","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-quantized-representations-for","slug":"hierarchical-quantized-representations-for","title":"Hierarchical Quantized Representations for Script Generation","date":"2018-08-28","arxiv_id":"1808.09542","repositories_listed":1,"syntology":null},{"url":"/paper/rational-recurrences","slug":"rational-recurrences","title":"Rational Recurrences","date":"2018-08-28","arxiv_id":"1808.09357","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rational-recurrences#ran","syntology_url":"https://syntology.ai/paper/1808.09357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09357"}},"official":{"repos":["Noahs-ARK/rational-recurrences"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/predefined-sparseness-in-recurrent-sequence","slug":"predefined-sparseness-in-recurrent-sequence","title":"Predefined Sparseness in Recurrent Sequence Models","date":"2018-08-27","arxiv_id":"1808.08720","repositories_listed":1,"syntology":null},{"url":"/paper/word-sense-induction-with-neural-bilm-and","slug":"word-sense-induction-with-neural-bilm-and","title":"Word Sense Induction with Neural biLM and Symmetric Patterns","date":"2018-08-26","arxiv_id":"1808.08518","repositories_listed":1,"syntology":null},{"url":"/paper/document-informed-neural-autoregressive-topic-1","slug":"document-informed-neural-autoregressive-topic-1","title":"Document Informed Neural Autoregressive Topic Models","date":"2018-08-11","arxiv_id":"1808.03793","repositories_listed":1,"syntology":null},{"url":"/paper/character-level-language-modeling-with-deeper","slug":"character-level-language-modeling-with-deeper","title":"Character-Level Language Modeling with Deeper Self-Attention","date":"2018-08-09","arxiv_id":"1808.04444","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-language-modeling-converging-on","slug":"large-scale-language-modeling-converging-on","title":"Large Scale Language Modeling: Converging on 40GB of Text in Four Hours","date":"2018-08-03","arxiv_id":"1808.01371","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-scale-language-modeling-converging-on#ran","syntology_url":"https://syntology.ai/paper/1808.01371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.01371"}},"official":{"repos":["NVIDIA/sentiment-discovery"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-string-embeddings-for-sequence","slug":"contextual-string-embeddings-for-sequence","title":"Contextual String Embeddings for Sequence Labeling","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-device-neural-language-model-based-word","slug":"on-device-neural-language-model-based-word","title":"On-Device Neural Language Model Based Word Prediction","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reproducing-and-regularizing-the-scrn-model","slug":"reproducing-and-regularizing-the-scrn-model","title":"Reproducing and Regularizing the SCRN Model","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rnn-simulations-of-grammaticality-judgments","slug":"rnn-simulations-of-grammaticality-judgments","title":"RNN Simulations of Grammaticality Judgments on Long-distance Dependencies","date":"2018-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-techniques-for-language-model","slug":"a-comparison-of-techniques-for-language-model","title":"A Comparison of Techniques for Language Model Integration in Encoder-Decoder Speech Recognition","date":"2018-07-27","arxiv_id":"1807.10857","repositories_listed":1,"syntology":null},{"url":"/paper/bilingual-expert-can-find-translation-errors","slug":"bilingual-expert-can-find-translation-errors","title":"\"Bilingual Expert\" Can Find Translation Errors","date":"2018-07-25","arxiv_id":"1807.09433","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-adaptation-techniques-and","slug":"a-comparison-of-adaptation-techniques-and","title":"A Comparison of Adaptation Techniques and Recurrent Neural Network Architectures","date":"2018-07-12","arxiv_id":"1807.06441","repositories_listed":1,"syntology":null},{"url":"/paper/deep-speare-a-joint-neural-model-of-poetic","slug":"deep-speare-a-joint-neural-model-of-poetic","title":"Deep-speare: A Joint Neural Model of Poetic Language, Meter and Rhyme","date":"2018-07-10","arxiv_id":"1807.03491","repositories_listed":1,"syntology":null},{"url":"/paper/improved-training-of-neural-trans-dimensional","slug":"improved-training-of-neural-trans-dimensional","title":"Improved training of neural trans-dimensional random field language models with dynamic noise-contrastive estimation","date":"2018-07-03","arxiv_id":"1807.00993","repositories_listed":1,"syntology":null},{"url":"/paper/baseline-a-library-for-rapid-modeling","slug":"baseline-a-library-for-rapid-modeling","title":"Baseline: A Library for Rapid Modeling, Experimentation and Development of Deep Learning Algorithms targeting NLP","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/document-modeling-with-external-attention-for","slug":"document-modeling-with-external-attention-for","title":"Document Modeling with External Attention for Sentence Extraction","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/subword-level-word-vector-representations-for","slug":"subword-level-word-vector-representations-for","title":"Subword-level Word Vector Representations for Korean","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/the-influence-of-context-on-sentence","slug":"the-influence-of-context-on-sentence","title":"The Influence of Context on Sentence Acceptability Judgements","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/word-error-rate-estimation-for-speech","slug":"word-error-rate-estimation-for-speech","title":"Word Error Rate Estimation for Speech Recognition: e-WER","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/handling-massive-n-gram-datasets-efficiently","slug":"handling-massive-n-gram-datasets-efficiently","title":"Handling Massive N-Gram Datasets Efficiently","date":"2018-06-25","arxiv_id":"1806.09447","repositories_listed":1,"syntology":null},{"url":"/paper/a-melody-conditioned-lyrics-language-model","slug":"a-melody-conditioned-lyrics-language-model","title":"A Melody-Conditioned Lyrics Language Model","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neural-sign-language-translation","slug":"neural-sign-language-translation","title":"Neural Sign Language Translation","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-text-style-transfer-using","slug":"unsupervised-text-style-transfer-using","title":"Unsupervised Text Style Transfer using Language Models as Discriminators","date":"2018-05-30","arxiv_id":"1805.11749","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-language-model-in-lstm-for-ocr","slug":"implicit-language-model-in-lstm-for-ocr","title":"Implicit Language Model in LSTM for OCR","date":"2018-05-23","arxiv_id":"1805.09441","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-cache-model-for-image-recognition-1","slug":"a-simple-cache-model-for-image-recognition-1","title":"A Simple Cache Model for Image Recognition","date":"2018-05-21","arxiv_id":"1805.08709","repositories_listed":1,"syntology":null},{"url":"/paper/character-based-neural-networks-for-sentence","slug":"character-based-neural-networks-for-sentence","title":"Character-based Neural Networks for Sentence Pair Modeling","date":"2018-05-21","arxiv_id":"1805.08297","repositories_listed":1,"syntology":null},{"url":"/paper/semstyle-learning-to-generate-stylised-image","slug":"semstyle-learning-to-generate-stylised-image","title":"SemStyle: Learning to Generate Stylised Image Captions using Unaligned Text","date":"2018-05-18","arxiv_id":"1805.07030","repositories_listed":1,"syntology":null}],"record_sha256":"f8eef37ab40c6a2bad61758dac1564e4fc288a2ba58095c67395b9d0d8e0159f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}