{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/wmt-2014/papers/ran/1","list_of":"/dataset/wmt-2014","dataset":"WMT 2014","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this dataset or check it against this dataset's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,40],"of":40,"counts":{"papers_with_a_benchmark_row":89,"with_a_code_link":81,"where_syntology_ran_a_sample":40,"not_listed_spam_title":0,"listed":89,"listed_where_code_ran":40,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":35,"every_run_a_failure_of_syntologys_instrument":5,"listed_with_a_run_with_no_instrument_failure":35,"listed_every_run_a_failure_of_syntologys_instrument":5,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/wmt-2014/papers/ran/1","prev":null,"next":null,"papers":[{"paper":"/paper/mega-moving-average-equipped-gated-attention","slug":"mega-moving-average-equipped-gated-attention","title":"Mega: Moving Average Equipped Gated Attention","date":"2022-09-21","arxiv_id":"2209.10655","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":12,"samples_constructed":6,"samples_ran_checked":8,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":11,"official":{"repos":["facebookresearch/mega"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mega-moving-average-equipped-gated-attention#ran","syntology_url":"https://syntology.ai/paper/2209.10655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10655"}}}},{"paper":"/paper/bert-mbert-or-bibert-a-study-on","slug":"bert-mbert-or-bibert-a-study-on","title":"BERT, mBERT, or BiBERT? A Study on Contextualized Embeddings for Neural Machine Translation","date":"2021-09-09","arxiv_id":"2109.04588","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":1,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["fe1ixxu/BiBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/bert-mbert-or-bibert-a-study-on#ran","syntology_url":"https://syntology.ai/paper/2109.04588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04588"}}}},{"paper":"/paper/r-drop-regularized-dropout-for-neural","slug":"r-drop-regularized-dropout-for-neural","title":"R-Drop: Regularized Dropout for Neural Networks","date":"2021-06-28","arxiv_id":"2106.14448","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":2,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["dropreg/R-Drop"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/r-drop-regularized-dropout-for-neural#ran","syntology_url":"https://syntology.ai/paper/2106.14448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14448"}}}},{"paper":"/paper/resmlp-feedforward-networks-for-image","slug":"resmlp-feedforward-networks-for-image","title":"ResMLP: Feedforward networks for image classification with data-efficient training","date":"2021-05-07","arxiv_id":"2105.03404","rows_on_this_dataset":4,"code_links":19,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["facebookresearch/deit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/resmlp-feedforward-networks-for-image#ran","syntology_url":"https://syntology.ai/paper/2105.03404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03404"}}}},{"paper":"/paper/lessons-on-parameter-sharing-across-layers-in","slug":"lessons-on-parameter-sharing-across-layers-in","title":"Lessons on Parameter Sharing across Layers in Transformers","date":"2021-04-13","arxiv_id":"2104.06022","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":2,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["takase/share_layer_params"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lessons-on-parameter-sharing-across-layers-in#ran","syntology_url":"https://syntology.ai/paper/2104.06022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06022"}}}},{"paper":"/paper/rethinking-perturbations-in-encoder-decoders","slug":"rethinking-perturbations-in-encoder-decoders","title":"Rethinking Perturbations in Encoder-Decoders for Fast Training","date":"2021-04-05","arxiv_id":"2104.01853","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["takase/rethink_perturbations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rethinking-perturbations-in-encoder-decoders#ran","syntology_url":"https://syntology.ai/paper/2104.01853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.01853"}}}},{"paper":"/paper/mask-attention-networks-rethinking-and","slug":"mask-attention-networks-rethinking-and","title":"Mask Attention Networks: Rethinking and Strengthen Transformer","date":"2021-03-25","arxiv_id":"2103.13597","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mask-attention-networks-rethinking-and#ran","syntology_url":"https://syntology.ai/paper/2103.13597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13597"}}}},{"paper":"/paper/finetuning-pretrained-transformers-into-rnns","slug":"finetuning-pretrained-transformers-into-rnns","title":"Finetuning Pretrained Transformers into RNNs","date":"2021-03-24","arxiv_id":"2103.13076","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":3,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/finetuning-pretrained-transformers-into-rnns#ran","syntology_url":"https://syntology.ai/paper/2103.13076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13076"}}}},{"paper":"/paper/incorporating-a-local-translation-mechanism","slug":"incorporating-a-local-translation-mechanism","title":"Incorporating a Local Translation Mechanism into Non-autoregressive Translation","date":"2020-11-12","arxiv_id":"2011.06132","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["shawnkx/NAT-with-Local-AT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/incorporating-a-local-translation-mechanism#ran","syntology_url":"https://syntology.ai/paper/2011.06132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.06132"}}}},{"paper":"/paper/very-deep-transformers-for-neural-machine","slug":"very-deep-transformers-for-neural-machine","title":"Very Deep Transformers for Neural Machine Translation","date":"2020-08-18","arxiv_id":"2008.07772","rows_on_this_dataset":3,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":8,"samples_constructed":2,"samples_ran_checked":3,"samples_ran_instrument_failed":5,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["namisan/exdeep-nmt"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/very-deep-transformers-for-neural-machine#ran","syntology_url":"https://syntology.ai/paper/2008.07772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.07772"}}}},{"paper":"/paper/multi-branch-attentive-transformer","slug":"multi-branch-attentive-transformer","title":"Multi-branch Attentive Transformer","date":"2020-06-18","arxiv_id":"2006.10270","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":2,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":3,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/multi-branch-attentive-transformer#ran","syntology_url":"https://syntology.ai/paper/2006.10270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.10270"}}}},{"paper":"/paper/language-models-are-few-shot-learners","slug":"language-models-are-few-shot-learners","title":"Language Models are Few-Shot Learners","date":"2020-05-28","arxiv_id":"2005.14165","rows_on_this_dataset":2,"code_links":67,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":65,"samples_ran":45,"samples_constructed":0,"samples_ran_checked":40,"samples_ran_instrument_failed":5,"samples_unverified":20,"pointer_only_for_licence":7,"official":{"repos":["openai/gpt-3"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/language-models-are-few-shot-learners#ran","syntology_url":"https://syntology.ai/paper/2005.14165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.14165"}}}},{"paper":"/paper/synthesizer-rethinking-self-attention-in","slug":"synthesizer-rethinking-self-attention-in","title":"Synthesizer: Rethinking Self-Attention in Transformer Models","date":"2020-05-02","arxiv_id":"2005.00743","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/synthesizer-rethinking-self-attention-in#ran","syntology_url":"https://syntology.ai/paper/2005.00743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00743"}}}},{"paper":"/paper/understanding-the-difficulty-of-training","slug":"understanding-the-difficulty-of-training","title":"Understanding the Difficulty of Training Transformers","date":"2020-04-17","arxiv_id":"2004.08249","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":2,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["LiyuanLucasLiu/Transforemr-Clinic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/understanding-the-difficulty-of-training#ran","syntology_url":"https://syntology.ai/paper/2004.08249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.08249"}}}},{"paper":"/paper/learning-to-encode-position-for-transformer","slug":"learning-to-encode-position-for-transformer","title":"Learning to Encode Position for Transformer with Continuous Dynamical Model","date":"2020-03-13","arxiv_id":"2003.09229","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":2,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":6,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/learning-to-encode-position-for-transformer#ran","syntology_url":"https://syntology.ai/paper/2003.09229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.09229"}}}},{"paper":"/paper/time-aware-large-kernel-convolutions","slug":"time-aware-large-kernel-convolutions","title":"Time-aware Large Kernel Convolutions","date":"2020-02-08","arxiv_id":"2002.03184","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["lioutasb/TaLKConvolutions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/time-aware-large-kernel-convolutions#ran","syntology_url":"https://syntology.ai/paper/2002.03184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03184"}}}},{"paper":"/paper/data-diversification-an-elegant-strategy-for","slug":"data-diversification-an-elegant-strategy-for","title":"Data Diversification: A Simple Strategy For Neural Machine Translation","date":"2019-11-05","arxiv_id":"1911.01986","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":4,"official":{"repos":["nxphi47/data_diversification"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/data-diversification-an-elegant-strategy-for#ran","syntology_url":"https://syntology.ai/paper/1911.01986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.01986"}}}},{"paper":"/paper/exploring-the-limits-of-transfer-learning","slug":"exploring-the-limits-of-transfer-learning","title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","date":"2019-10-23","arxiv_id":"1910.10683","rows_on_this_dataset":2,"code_links":57,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":31,"samples_ran":21,"samples_constructed":0,"samples_ran_checked":20,"samples_ran_instrument_failed":1,"samples_unverified":10,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/exploring-the-limits-of-transfer-learning#ran","syntology_url":"https://syntology.ai/paper/1910.10683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.10683"}}}},{"paper":"/paper/flowseq-non-autoregressive-conditional","slug":"flowseq-non-autoregressive-conditional","title":"FlowSeq: Non-Autoregressive Conditional Sequence Generation with Generative Flow","date":"2019-09-05","arxiv_id":"1909.02480","rows_on_this_dataset":10,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["XuezheMax/flowseq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/flowseq-non-autoregressive-conditional#ran","syntology_url":"https://syntology.ai/paper/1909.02480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02480"}}}},{"paper":"/paper/adaptively-sparse-transformers","slug":"adaptively-sparse-transformers","title":"Adaptively Sparse Transformers","date":"2019-08-30","arxiv_id":"1909.00015","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["deep-spin/entmax"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/adaptively-sparse-transformers#ran","syntology_url":"https://syntology.ai/paper/1909.00015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.00015"}}}},{"paper":"/paper/cross-lingual-language-model-pretraining","slug":"cross-lingual-language-model-pretraining","title":"Cross-lingual Language Model Pretraining","date":"2019-01-22","arxiv_id":"1901.07291","rows_on_this_dataset":2,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":1,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cross-lingual-language-model-pretraining#ran","syntology_url":"https://syntology.ai/paper/1901.07291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.07291"}}}},{"paper":"/paper/unsupervised-neural-machine-translation-with","slug":"unsupervised-neural-machine-translation-with","title":"Unsupervised Neural Machine Translation with SMT as Posterior Regularization","date":"2019-01-14","arxiv_id":"1901.04112","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":8,"pointer_only_for_licence":0,"official":{"repos":["Imagist-Shuo/UNMT-SPR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/unsupervised-neural-machine-translation-with#ran","syntology_url":"https://syntology.ai/paper/1901.04112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.04112"}}}},{"paper":"/paper/understanding-back-translation-at-scale","slug":"understanding-back-translation-at-scale","title":"Understanding Back-Translation at Scale","date":"2018-08-28","arxiv_id":"1808.09381","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/understanding-back-translation-at-scale#ran","syntology_url":"https://syntology.ai/paper/1808.09381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09381"}}}},{"paper":"/paper/universal-transformers","slug":"universal-transformers","title":"Universal Transformers","date":"2018-07-10","arxiv_id":"1807.03819","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":25,"samples_ran":17,"samples_constructed":8,"samples_ran_checked":17,"samples_ran_instrument_failed":0,"samples_unverified":8,"pointer_only_for_licence":24,"official":{"repos":["tensorflow/tensor2tensor"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/universal-transformers#ran","syntology_url":"https://syntology.ai/paper/1807.03819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03819"}}}},{"paper":"/paper/accelerating-neural-transformer-via-an","slug":"accelerating-neural-transformer-via-an","title":"Accelerating Neural Transformer via an Average Attention Network","date":"2018-05-02","arxiv_id":"1805.00631","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":19,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":12,"pointer_only_for_licence":0,"official":{"repos":["bzhangXMU/transformer-aan"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":12,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/accelerating-neural-transformer-via-an#ran","syntology_url":"https://syntology.ai/paper/1805.00631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.00631"}}}},{"paper":"/paper/self-attention-with-relative-position","slug":"self-attention-with-relative-position","title":"Self-Attention with Relative Position Representations","date":"2018-03-06","arxiv_id":"1803.02155","rows_on_this_dataset":2,"code_links":13,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":23,"samples_ran":13,"samples_constructed":4,"samples_ran_checked":8,"samples_ran_instrument_failed":5,"samples_unverified":10,"pointer_only_for_licence":3,"official":{"repos":["tensorflow/tensor2tensor"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-attention-with-relative-position#ran","syntology_url":"https://syntology.ai/paper/1803.02155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.02155"}}}},{"paper":"/paper/deterministic-non-autoregressive-neural","slug":"deterministic-non-autoregressive-neural","title":"Deterministic Non-Autoregressive Neural Sequence Modeling by Iterative Refinement","date":"2018-02-19","arxiv_id":"1802.06901","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["nyu-dl/dl4mt-nonauto"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deterministic-non-autoregressive-neural#ran","syntology_url":"https://syntology.ai/paper/1802.06901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.06901"}}}},{"paper":"/paper/non-autoregressive-neural-machine-translation-1","slug":"non-autoregressive-neural-machine-translation-1","title":"Non-Autoregressive Neural Machine Translation","date":"2017-11-07","arxiv_id":"1711.02281","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["salesforce/nonauto-nmt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/non-autoregressive-neural-machine-translation-1#ran","syntology_url":"https://syntology.ai/paper/1711.02281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.02281"}}}},{"paper":"/paper/weighted-transformer-network-for-machine","slug":"weighted-transformer-network-for-machine","title":"Weighted Transformer Network for Machine Translation","date":"2017-11-06","arxiv_id":"1711.02132","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/weighted-transformer-network-for-machine#ran","syntology_url":"https://syntology.ai/paper/1711.02132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.02132"}}}},{"paper":"/paper/unsupervised-neural-machine-translation","slug":"unsupervised-neural-machine-translation","title":"Unsupervised Neural Machine Translation","date":"2017-10-30","arxiv_id":"1710.11041","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":5,"samples_unverified":0,"pointer_only_for_licence":6,"official":{"repos":["artetxem/undreamt","rsennrich/subword-nmt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/unsupervised-neural-machine-translation#ran","syntology_url":"https://syntology.ai/paper/1710.11041","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11041"}}}},{"paper":"/paper/simple-recurrent-units-for-highly","slug":"simple-recurrent-units-for-highly","title":"Simple Recurrent Units for Highly Parallelizable Recurrence","date":"2017-09-08","arxiv_id":"1709.02755","rows_on_this_dataset":1,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["asappresearch/sru"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/simple-recurrent-units-for-highly#ran","syntology_url":"https://syntology.ai/paper/1709.02755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.02755"}}}},{"paper":"/paper/attention-is-all-you-need","slug":"attention-is-all-you-need","title":"Attention Is All You Need","date":"2017-06-12","arxiv_id":"1706.03762","rows_on_this_dataset":4,"code_links":595,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":946,"samples_ran":610,"samples_constructed":293,"samples_ran_checked":529,"samples_ran_instrument_failed":81,"samples_unverified":336,"pointer_only_for_licence":451,"official":{"repos":["tensorflow/tensor2tensor"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/attention-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/1706.03762","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1706.03762"}}}},{"paper":"/paper/outrageously-large-neural-networks-the","slug":"outrageously-large-neural-networks-the","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","date":"2017-01-23","arxiv_id":"1701.06538","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":4,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":6,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/outrageously-large-neural-networks-the#ran","syntology_url":"https://syntology.ai/paper/1701.06538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1701.06538"}}}},{"paper":"/paper/googles-neural-machine-translation-system","slug":"googles-neural-machine-translation-system","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","date":"2016-09-26","arxiv_id":"1609.08144","rows_on_this_dataset":2,"code_links":28,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":46,"samples_ran":34,"samples_constructed":4,"samples_ran_checked":26,"samples_ran_instrument_failed":8,"samples_unverified":12,"pointer_only_for_licence":13,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/googles-neural-machine-translation-system#ran","syntology_url":"https://syntology.ai/paper/1609.08144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.08144"}}}},{"paper":"/paper/sequence-level-knowledge-distillation","slug":"sequence-level-knowledge-distillation","title":"Sequence-Level Knowledge Distillation","date":"2016-06-25","arxiv_id":"1606.07947","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["harvardnlp/nmt-android","harvardnlp/seq2seq-attn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sequence-level-knowledge-distillation#ran","syntology_url":"https://syntology.ai/paper/1606.07947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.07947"}}}},{"paper":"/paper/effective-approaches-to-attention-based","slug":"effective-approaches-to-attention-based","title":"Effective Approaches to Attention-based Neural Machine Translation","date":"2015-08-17","arxiv_id":"1508.04025","rows_on_this_dataset":3,"code_links":44,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/effective-approaches-to-attention-based#ran","syntology_url":"https://syntology.ai/paper/1508.04025","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1508.04025"}}}},{"paper":"/paper/sequence-to-sequence-learning-with-neural","slug":"sequence-to-sequence-learning-with-neural","title":"Sequence to Sequence Learning with Neural Networks","date":"2014-09-10","arxiv_id":"1409.3215","rows_on_this_dataset":2,"code_links":74,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":25,"samples_ran":21,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":8,"samples_unverified":4,"pointer_only_for_licence":9,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sequence-to-sequence-learning-with-neural#ran","syntology_url":"https://syntology.ai/paper/1409.3215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1409.3215"}}}},{"paper":"/paper/recurrent-neural-network-regularization","slug":"recurrent-neural-network-regularization","title":"Recurrent Neural Network Regularization","date":"2014-09-08","arxiv_id":"1409.2329","rows_on_this_dataset":1,"code_links":21,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":4,"pointer_only_for_licence":6,"official":{"repos":["wojzaremba/lstm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/recurrent-neural-network-regularization#ran","syntology_url":"https://syntology.ai/paper/1409.2329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1409.2329"}}}},{"paper":"/paper/neural-machine-translation-by-jointly","slug":"neural-machine-translation-by-jointly","title":"Neural Machine Translation by Jointly Learning to Align and Translate","date":"2014-09-01","arxiv_id":"1409.0473","rows_on_this_dataset":1,"code_links":124,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":44,"samples_ran":41,"samples_constructed":2,"samples_ran_checked":24,"samples_ran_instrument_failed":17,"samples_unverified":3,"pointer_only_for_licence":16,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/neural-machine-translation-by-jointly#ran","syntology_url":"https://syntology.ai/paper/1409.0473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1409.0473"}}}},{"paper":"/paper/learning-phrase-representations-using-rnn","slug":"learning-phrase-representations-using-rnn","title":"Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation","date":"2014-06-03","arxiv_id":"1406.1078","rows_on_this_dataset":1,"code_links":42,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":22,"samples_ran":13,"samples_constructed":6,"samples_ran_checked":10,"samples_ran_instrument_failed":3,"samples_unverified":9,"pointer_only_for_licence":14,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/learning-phrase-representations-using-rnn#ran","syntology_url":"https://syntology.ai/paper/1406.1078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1406.1078"}}}}],"record_sha256":"383bc8905263e4c132fd676f6e2fa1c3ce7d516808bbc7052fd199683e6aefea","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}