{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/wikitext-103/papers/ran/1","list_of":"/dataset/wikitext-103","dataset":"WikiText-103","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this dataset or check it against this dataset's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,31],"of":31,"counts":{"papers_with_a_benchmark_row":55,"with_a_code_link":51,"where_syntology_ran_a_sample":31,"not_listed_spam_title":0,"listed":55,"listed_where_code_ran":31,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":28,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":28,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/wikitext-103/papers/ran/1","prev":null,"next":null,"papers":[{"paper":"/paper/gateloop-fully-data-controlled-linear","slug":"gateloop-fully-data-controlled-linear","title":"GateLoop: Fully Data-Controlled Linear Recurrence for Sequence Modeling","date":"2023-11-03","arxiv_id":"2311.01927","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["tobiaskatsch/GateLoop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/gateloop-fully-data-controlled-linear#ran","syntology_url":"https://syntology.ai/paper/2311.01927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01927"}}}},{"paper":"/paper/primal-attention-self-attention-through","slug":"primal-attention-self-attention-through","title":"Primal-Attention: Self-attention through Asymmetric Kernel SVD in Primal Representation","date":"2023-05-31","arxiv_id":"2305.19798","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["yingyichen-cyy/PrimalAttention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/primal-attention-self-attention-through#ran","syntology_url":"https://syntology.ai/paper/2305.19798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.19798"}}}},{"paper":"/paper/hyena-hierarchy-towards-larger-convolutional","slug":"hyena-hierarchy-towards-larger-convolutional","title":"Hyena Hierarchy: Towards Larger Convolutional Language Models","date":"2023-02-21","arxiv_id":"2302.10866","rows_on_this_dataset":2,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":5,"samples_unverified":0,"pointer_only_for_licence":4,"official":{"repos":["hazyresearch/safari"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hyena-hierarchy-towards-larger-convolutional#ran","syntology_url":"https://syntology.ai/paper/2302.10866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10866"}}}},{"paper":"/paper/hungry-hungry-hippos-towards-language","slug":"hungry-hungry-hippos-towards-language","title":"Hungry Hungry Hippos: Towards Language Modeling with State Space Models","date":"2022-12-28","arxiv_id":"2212.14052","rows_on_this_dataset":5,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":4,"samples_unverified":8,"pointer_only_for_licence":0,"official":{"repos":["hazyresearch/h3"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hungry-hungry-hippos-towards-language#ran","syntology_url":"https://syntology.ai/paper/2212.14052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.14052"}}}},{"paper":"/paper/you-can-t-pick-your-neighbors-or-can-you-when","slug":"you-can-t-pick-your-neighbors-or-can-you-when","title":"You can't pick your neighbors, or can you? When and how to rely on retrieval in the $k$NN-LM","date":"2022-10-28","arxiv_id":"2210.15859","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["iesl/knnlm-retrieval-quality"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/you-can-t-pick-your-neighbors-or-can-you-when#ran","syntology_url":"https://syntology.ai/paper/2210.15859","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.15859"}}}},{"paper":"/paper/mega-moving-average-equipped-gated-attention","slug":"mega-moving-average-equipped-gated-attention","title":"Mega: Moving Average Equipped Gated Attention","date":"2022-09-21","arxiv_id":"2209.10655","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":12,"samples_constructed":6,"samples_ran_checked":8,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":11,"official":{"repos":["facebookresearch/mega"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mega-moving-average-equipped-gated-attention#ran","syntology_url":"https://syntology.ai/paper/2209.10655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10655"}}}},{"paper":"/paper/general-purpose-long-context-autoregressive","slug":"general-purpose-long-context-autoregressive","title":"General-purpose, long-context autoregressive modeling with Perceiver AR","date":"2022-02-15","arxiv_id":"2202.07765","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":4,"samples_unverified":3,"pointer_only_for_licence":1,"official":{"repos":["google-research/perceiver-ar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/general-purpose-long-context-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2202.07765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.07765"}}}},{"paper":"/paper/improving-language-models-by-retrieving-from","slug":"improving-language-models-by-retrieving-from","title":"Improving language models by retrieving from trillions of tokens","date":"2021-12-08","arxiv_id":"2112.04426","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":23,"samples_ran":16,"samples_constructed":5,"samples_ran_checked":14,"samples_ran_instrument_failed":2,"samples_unverified":7,"pointer_only_for_licence":3,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/improving-language-models-by-retrieving-from#ran","syntology_url":"https://syntology.ai/paper/2112.04426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.04426"}}}},{"paper":"/paper/efficiently-modeling-long-sequences-with-1","slug":"efficiently-modeling-long-sequences-with-1","title":"Efficiently Modeling Long Sequences with Structured State Spaces","date":"2021-10-31","arxiv_id":"2111.00396","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":55,"samples_ran":37,"samples_constructed":8,"samples_ran_checked":21,"samples_ran_instrument_failed":16,"samples_unverified":18,"pointer_only_for_licence":3,"official":{"repos":["state-spaces/s4"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["community","listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficiently-modeling-long-sequences-with-1#ran","syntology_url":"https://syntology.ai/paper/2111.00396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.00396"}}}},{"paper":"/paper/revisiting-simple-neural-probabilistic","slug":"revisiting-simple-neural-probabilistic","title":"Revisiting Simple Neural Probabilistic Language Models","date":"2021-04-08","arxiv_id":"2104.03474","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["SimengSun/revisit-nplm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/revisiting-simple-neural-probabilistic#ran","syntology_url":"https://syntology.ai/paper/2104.03474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03474"}}}},{"paper":"/paper/finetuning-pretrained-transformers-into-rnns","slug":"finetuning-pretrained-transformers-into-rnns","title":"Finetuning Pretrained Transformers into RNNs","date":"2021-03-24","arxiv_id":"2103.13076","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":3,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":5,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/finetuning-pretrained-transformers-into-rnns#ran","syntology_url":"https://syntology.ai/paper/2103.13076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.13076"}}}},{"paper":"/paper/all-nlp-tasks-are-generation-tasks-a-general","slug":"all-nlp-tasks-are-generation-tasks-a-general","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-03-18","arxiv_id":"2103.10360","rows_on_this_dataset":2,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["THUDM/GLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/all-nlp-tasks-are-generation-tasks-a-general#ran","syntology_url":"https://syntology.ai/paper/2103.10360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.10360"}}}},{"paper":"/paper/rethinking-attention-with-performers","slug":"rethinking-attention-with-performers","title":"Rethinking Attention with Performers","date":"2020-09-30","arxiv_id":"2009.14794","rows_on_this_dataset":1,"code_links":7,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":16,"samples_ran":11,"samples_constructed":3,"samples_ran_checked":7,"samples_ran_instrument_failed":4,"samples_unverified":5,"pointer_only_for_licence":6,"official":{"repos":["google-research/google-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/rethinking-attention-with-performers#ran","syntology_url":"https://syntology.ai/paper/2009.14794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14794"}}}},{"paper":"/paper/delight-very-deep-and-light-weight","slug":"delight-very-deep-and-light-weight","title":"DeLighT: Deep and Light-weight Transformer","date":"2020-08-03","arxiv_id":"2008.00623","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["sacmehta/delight"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/delight-very-deep-and-light-weight#ran","syntology_url":"https://syntology.ai/paper/2008.00623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.00623"}}}},{"paper":"/paper/transformers-are-rnns-fast-autoregressive","slug":"transformers-are-rnns-fast-autoregressive","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","date":"2020-06-29","arxiv_id":"2006.16236","rows_on_this_dataset":1,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["idiap/fast-transformers"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/transformers-are-rnns-fast-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2006.16236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16236"}}}},{"paper":"/paper/efficient-content-based-sparse-attention-with-1","slug":"efficient-content-based-sparse-attention-with-1","title":"Efficient Content-Based Sparse Attention with Routing Transformers","date":"2020-03-12","arxiv_id":"2003.05997","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":2,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/efficient-content-based-sparse-attention-with-1#ran","syntology_url":"https://syntology.ai/paper/2003.05997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.05997"}}}},{"paper":"/paper/accessing-higher-level-representations-in","slug":"accessing-higher-level-representations-in","title":"Addressing Some Limitations of Transformers with Feedback Memory","date":"2020-02-21","arxiv_id":"2002.09402","rows_on_this_dataset":2,"code_links":4,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["facebookresearch/transformer-sequential"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/accessing-higher-level-representations-in#ran","syntology_url":"https://syntology.ai/paper/2002.09402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09402"}}}},{"paper":"/paper/time-aware-large-kernel-convolutions","slug":"time-aware-large-kernel-convolutions","title":"Time-aware Large Kernel Convolutions","date":"2020-02-08","arxiv_id":"2002.03184","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":2,"samples_ran_checked":2,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["lioutasb/TaLKConvolutions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/time-aware-large-kernel-convolutions#ran","syntology_url":"https://syntology.ai/paper/2002.03184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03184"}}}},{"paper":"/paper/reformer-the-efficient-transformer-1","slug":"reformer-the-efficient-transformer-1","title":"Reformer: The Efficient Transformer","date":"2020-01-13","arxiv_id":"2001.04451","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":1,"samples_ran_checked":1,"samples_ran_instrument_failed":5,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["google/trax"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/reformer-the-efficient-transformer-1#ran","syntology_url":"https://syntology.ai/paper/2001.04451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.04451"}}}},{"paper":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":5,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/compressive-transformers-for-long-range-1#ran","syntology_url":"https://syntology.ai/paper/1911.05507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.05507"}}}},{"paper":"/paper/generalization-through-memorization-nearest","slug":"generalization-through-memorization-nearest","title":"Generalization through Memorization: Nearest Neighbor Language Models","date":"2019-11-01","arxiv_id":"1911.00172","rows_on_this_dataset":2,"code_links":5,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["urvashik/knnlm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/generalization-through-memorization-nearest#ran","syntology_url":"https://syntology.ai/paper/1911.00172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.00172"}}}},{"paper":"/paper/on-the-adequacy-of-untuned-warmup-for","slug":"on-the-adequacy-of-untuned-warmup-for","title":"On the adequacy of untuned warmup for adaptive optimization","date":"2019-10-09","arxiv_id":"1910.04209","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/on-the-adequacy-of-untuned-warmup-for#ran","syntology_url":"https://syntology.ai/paper/1910.04209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.04209"}}}},{"paper":"/paper/megatron-lm-training-multi-billion-parameter","slug":"megatron-lm-training-multi-billion-parameter","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","date":"2019-09-17","arxiv_id":"1909.08053","rows_on_this_dataset":1,"code_links":10,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":47,"samples_ran":12,"samples_constructed":2,"samples_ran_checked":7,"samples_ran_instrument_failed":5,"samples_unverified":35,"pointer_only_for_licence":15,"official":{"repos":["NVIDIA/Megatron-LM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/megatron-lm-training-multi-billion-parameter#ran","syntology_url":"https://syntology.ai/paper/1909.08053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.08053"}}}},{"paper":"/paper/deep-equilibrium-models","slug":"deep-equilibrium-models","title":"Deep Equilibrium Models","date":"2019-09-03","arxiv_id":"1909.01377","rows_on_this_dataset":3,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":4,"official":{"repos":["locuslab/deq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deep-equilibrium-models#ran","syntology_url":"https://syntology.ai/paper/1909.01377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.01377"}}}},{"paper":"/paper/augmenting-self-attention-with-persistent","slug":"augmenting-self-attention-with-persistent","title":"Augmenting Self-attention with Persistent Memory","date":"2019-07-02","arxiv_id":"1907.01470","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":4,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/augmenting-self-attention-with-persistent#ran","syntology_url":"https://syntology.ai/paper/1907.01470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.01470"}}}},{"paper":"/paper/190409408","slug":"190409408","title":"Language Models with Transformers","date":"2019-04-20","arxiv_id":"1904.09408","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["cgraywang/gluon-nlp-1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/190409408#ran","syntology_url":"https://syntology.ai/paper/1904.09408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09408"}}}},{"paper":"/paper/transformer-xl-attentive-language-models","slug":"transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","arxiv_id":"1901.02860","rows_on_this_dataset":2,"code_links":37,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":143,"samples_ran":67,"samples_constructed":37,"samples_ran_checked":49,"samples_ran_instrument_failed":18,"samples_unverified":76,"pointer_only_for_licence":43,"official":{"repos":["kimiyoung/transformer-xl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["listed","official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/transformer-xl-attentive-language-models#ran","syntology_url":"https://syntology.ai/paper/1901.02860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.02860"}}}},{"paper":"/paper/trellis-networks-for-sequence-modeling","slug":"trellis-networks-for-sequence-modeling","title":"Trellis Networks for Sequence Modeling","date":"2018-10-15","arxiv_id":"1810.06682","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":2,"official":{"repos":["locuslab/trellisnet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/trellis-networks-for-sequence-modeling#ran","syntology_url":"https://syntology.ai/paper/1810.06682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.06682"}}}},{"paper":"/paper/relational-recurrent-neural-networks","slug":"relational-recurrent-neural-networks","title":"Relational recurrent neural networks","date":"2018-06-05","arxiv_id":"1806.01822","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/relational-recurrent-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1806.01822","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.01822"}}}},{"paper":"/paper/an-analysis-of-neural-language-modeling-at","slug":"an-analysis-of-neural-language-modeling-at","title":"An Analysis of Neural Language Modeling at Multiple Scales","date":"2018-03-22","arxiv_id":"1803.08240","rows_on_this_dataset":1,"code_links":12,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":14,"samples_constructed":0,"samples_ran_checked":13,"samples_ran_instrument_failed":1,"samples_unverified":4,"pointer_only_for_licence":4,"official":{"repos":["salesforce/awd-lstm-lm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-analysis-of-neural-language-modeling-at#ran","syntology_url":"https://syntology.ai/paper/1803.08240","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.08240"}}}},{"paper":"/paper/an-empirical-evaluation-of-generic","slug":"an-empirical-evaluation-of-generic","title":"An Empirical Evaluation of Generic Convolutional and Recurrent Networks for Sequence Modeling","date":"2018-03-04","arxiv_id":"1803.01271","rows_on_this_dataset":1,"code_links":35,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["locuslab/TCN"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-empirical-evaluation-of-generic#ran","syntology_url":"https://syntology.ai/paper/1803.01271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1803.01271"}}}}],"record_sha256":"ec1684e1ac6b77955cb2b2e5c981e6ff47b1532f83dcc350a01c99d1cbccf7bd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}