{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modelling/papers/2","list_of":"/task/language-modelling","task":"Language Modelling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":177,"rows_per_page":100,"rows":[101,200],"of":17610,"counts":{"archive_papers_tagged":17610,"with_a_code_link":7012,"where_syntology_ran_a_sample":2428,"not_listed_spam_title":0,"listed":17610,"listed_where_code_ran":2428,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":2027,"every_run_a_failure_of_syntologys_instrument":401,"listed_with_a_run_with_no_instrument_failure":2027,"listed_every_run_a_failure_of_syntologys_instrument":401,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modelling","prev":"/task/language-modelling","next":"/task/language-modelling/papers/3","papers":[{"url":"/paper/multitask-prompted-training-enables-zero-shot-1","slug":"multitask-prompted-training-enables-zero-shot-1","title":"Multitask Prompted Training Enables Zero-Shot Task Generalization","date":"2021-10-15","arxiv_id":"2110.08207","repositories_listed":8,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multitask-prompted-training-enables-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2110.08207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.08207"}},"official":{"repos":["bigscience-workshop/promptsource","bigscience-workshop/t-zero"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trocr-transformer-based-optical-character","slug":"trocr-transformer-based-optical-character","title":"TrOCR: Transformer-based Optical Character Recognition with Pre-trained Models","date":"2021-09-21","arxiv_id":"2109.10282","repositories_listed":8,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trocr-transformer-based-optical-character#ran","syntology_url":"https://syntology.ai/paper/2109.10282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10282"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/finetuned-language-models-are-zero-shot","slug":"finetuned-language-models-are-zero-shot","title":"Finetuned Language Models Are Zero-Shot Learners","date":"2021-09-03","arxiv_id":"2109.01652","repositories_listed":8,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/finetuned-language-models-are-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2109.01652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.01652"}},"official":{"repos":["google-research/flan"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/r-drop-regularized-dropout-for-neural","slug":"r-drop-regularized-dropout-for-neural","title":"R-Drop: Regularized Dropout for Neural Networks","date":"2021-06-28","arxiv_id":"2106.14448","repositories_listed":8,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/r-drop-regularized-dropout-for-neural#ran","syntology_url":"https://syntology.ai/paper/2106.14448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.14448"}},"official":{"repos":["dropreg/R-Drop"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/all-nlp-tasks-are-generation-tasks-a-general","slug":"all-nlp-tasks-are-generation-tasks-a-general","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-03-18","arxiv_id":"2103.10360","repositories_listed":8,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/all-nlp-tasks-are-generation-tasks-a-general#ran","syntology_url":"https://syntology.ai/paper/2103.10360","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.10360"}},"official":{"repos":["THUDM/GLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/switch-transformers-scaling-to-trillion","slug":"switch-transformers-scaling-to-trillion","title":"Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity","date":"2021-01-11","arxiv_id":"2101.03961","repositories_listed":8,"syntology":{"n":13,"n_ran":7,"n_constructed":2,"n_ran_checked":3,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/switch-transformers-scaling-to-trillion#ran","syntology_url":"https://syntology.ai/paper/2101.03961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.03961"}},"official":{"repos":["tensorflow/mesh"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/adabelief-optimizer-adapting-stepsizes-by-the","slug":"adabelief-optimizer-adapting-stepsizes-by-the","title":"AdaBelief Optimizer: Adapting Stepsizes by the Belief in Observed Gradients","date":"2020-10-15","arxiv_id":"2010.07468","repositories_listed":8,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adabelief-optimizer-adapting-stepsizes-by-the#ran","syntology_url":"https://syntology.ai/paper/2010.07468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07468"}},"official":{"repos":["juntang-zhuang/Adabelief-Optimizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transformers-are-rnns-fast-autoregressive","slug":"transformers-are-rnns-fast-autoregressive","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","date":"2020-06-29","arxiv_id":"2006.16236","repositories_listed":8,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/transformers-are-rnns-fast-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2006.16236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16236"}},"official":{"repos":["idiap/fast-transformers"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/camembert-a-tasty-french-language-model","slug":"camembert-a-tasty-french-language-model","title":"CamemBERT: a Tasty French Language Model","date":"2019-11-10","arxiv_id":"1911.03894","repositories_listed":8,"syntology":null},{"url":"/paper/ctrl-a-conditional-transformer-language-model-1","slug":"ctrl-a-conditional-transformer-language-model-1","title":"CTRL: A Conditional Transformer Language Model for Controllable Generation","date":"2019-09-11","arxiv_id":"1909.05858","repositories_listed":8,"syntology":{"n":14,"n_ran":14,"n_constructed":3,"n_ran_checked":11,"n_instrument":3,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"14 ran (of which 3 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ctrl-a-conditional-transformer-language-model-1#ran","syntology_url":"https://syntology.ai/paper/1909.05858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.05858"}},"official":null}},{"url":"/paper/translating-mathematical-formula-images-to","slug":"translating-mathematical-formula-images-to","title":"Translating Math Formula Images to LaTeX Sequences Using Deep Neural Networks with Sequence-level Training","date":"2019-08-29","arxiv_id":"1908.11415","repositories_listed":8,"syntology":null},{"url":"/paper/adaptive-attention-span-in-transformers","slug":"adaptive-attention-span-in-transformers","title":"Adaptive Attention Span in Transformers","date":"2019-05-19","arxiv_id":"1905.07799","repositories_listed":8,"syntology":null},{"url":"/paper/universal-transformers","slug":"universal-transformers","title":"Universal Transformers","date":"2018-07-10","arxiv_id":"1807.03819","repositories_listed":8,"syntology":{"n":25,"n_ran":17,"n_constructed":8,"n_ran_checked":17,"n_instrument":0,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":16,"n_pointer_only":24,"phrase":"17 ran (of which 8 constructed an object rather than computing a result; 17 with no instrument failure: 1 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/universal-transformers#ran","syntology_url":"https://syntology.ai/paper/1807.03819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.03819"}},"official":{"repos":["tensorflow/tensor2tensor"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/2309-06180","slug":"2309-06180","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","date":"2023-09-12","arxiv_id":"2309.06180","repositories_listed":7,"syntology":{"n":36,"n_ran":28,"n_constructed":0,"n_ran_checked":22,"n_instrument":6,"n_unverified":8,"n_honours":3,"n_violates":1,"n_no_contract":18,"n_pointer_only":11,"phrase":"28 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 3 honoured, 1 violated, 18 with no contract checked; 6 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/2309-06180#ran","syntology_url":"https://syntology.ai/paper/2309.06180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06180"}},"official":{"repos":["vllm-project/vllm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/extending-context-window-of-large-language","slug":"extending-context-window-of-large-language","title":"Extending Context Window of Large Language Models via Positional Interpolation","date":"2023-06-27","arxiv_id":"2306.15595","repositories_listed":7,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":6,"n_instrument":7,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 7 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/extending-context-window-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2306.15595","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15595"}},"official":null}},{"url":"/paper/sophia-a-scalable-stochastic-second-order","slug":"sophia-a-scalable-stochastic-second-order","title":"Sophia: A Scalable Stochastic Second-order Optimizer for Language Model Pre-training","date":"2023-05-23","arxiv_id":"2305.14342","repositories_listed":7,"syntology":{"n":19,"n_ran":13,"n_constructed":4,"n_ran_checked":12,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sophia-a-scalable-stochastic-second-order#ran","syntology_url":"https://syntology.ai/paper/2305.14342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14342"}},"official":null}},{"url":"/paper/nlg-evaluation-metrics-beyond-correlation","slug":"nlg-evaluation-metrics-beyond-correlation","title":"NLG Evaluation Metrics Beyond Correlation Analysis: An Empirical Metric Preference Checklist","date":"2023-05-15","arxiv_id":"2305.08566","repositories_listed":7,"syntology":null},{"url":"/paper/llama-adapter-efficient-fine-tuning-of","slug":"llama-adapter-efficient-fine-tuning-of","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","date":"2023-03-28","arxiv_id":"2303.16199","repositories_listed":7,"syntology":null},{"url":"/paper/hyena-hierarchy-towards-larger-convolutional","slug":"hyena-hierarchy-towards-larger-convolutional","title":"Hyena Hierarchy: Towards Larger Convolutional Language Models","date":"2023-02-21","arxiv_id":"2302.10866","repositories_listed":7,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hyena-hierarchy-towards-larger-convolutional#ran","syntology_url":"https://syntology.ai/paper/2302.10866","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.10866"}},"official":{"repos":["hazyresearch/safari"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/neural-codec-language-models-are-zero-shot","slug":"neural-codec-language-models-are-zero-shot","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","date":"2023-01-05","arxiv_id":"2301.02111","repositories_listed":7,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-codec-language-models-are-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2301.02111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.02111"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fast-inference-from-transformers-via","slug":"fast-inference-from-transformers-via","title":"Fast Inference from Transformers via Speculative Decoding","date":"2022-11-30","arxiv_id":"2211.17192","repositories_listed":7,"syntology":{"n":21,"n_ran":15,"n_constructed":1,"n_ran_checked":6,"n_instrument":9,"n_unverified":6,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"15 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 9 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fast-inference-from-transformers-via#ran","syntology_url":"https://syntology.ai/paper/2211.17192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.17192"}},"official":null}},{"url":"/paper/bloom-a-176b-parameter-open-access","slug":"bloom-a-176b-parameter-open-access","title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model","date":"2022-11-09","arxiv_id":"2211.05100","repositories_listed":7,"syntology":{"n":11,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/bloom-a-176b-parameter-open-access#ran","syntology_url":"https://syntology.ai/paper/2211.05100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.05100"}},"official":null}},{"url":"/paper/interpretability-in-the-wild-a-circuit-for","slug":"interpretability-in-the-wild-a-circuit-for","title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","date":"2022-11-01","arxiv_id":"2211.00593","repositories_listed":7,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/interpretability-in-the-wild-a-circuit-for#ran","syntology_url":"https://syntology.ai/paper/2211.00593","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.00593"}},"official":{"repos":["redwoodresearch/easy-transformer"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mega-moving-average-equipped-gated-attention","slug":"mega-moving-average-equipped-gated-attention","title":"Mega: Moving Average Equipped Gated Attention","date":"2022-09-21","arxiv_id":"2209.10655","repositories_listed":7,"syntology":{"n":13,"n_ran":12,"n_constructed":6,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"12 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mega-moving-average-equipped-gated-attention#ran","syntology_url":"https://syntology.ai/paper/2209.10655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10655"}},"official":{"repos":["facebookresearch/mega"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","slug":"palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":32,"n_constructed":16,"n_ran_checked":24,"n_instrument":8,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":21,"n_pointer_only":0,"phrase":"32 ran (of which 16 constructed an object rather than computing a result; 24 with no instrument failure: 2 honoured, 1 violated, 21 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/palm-scaling-language-modeling-with-pathways-1#ran","syntology_url":"https://syntology.ai/paper/2204.02311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02311"}},"official":null}},{"url":"/paper/rethinking-attention-with-performers","slug":"rethinking-attention-with-performers","title":"Rethinking Attention with Performers","date":"2020-09-30","arxiv_id":"2009.14794","repositories_listed":7,"syntology":{"n":16,"n_ran":11,"n_constructed":3,"n_ran_checked":7,"n_instrument":4,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"11 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rethinking-attention-with-performers#ran","syntology_url":"https://syntology.ai/paper/2009.14794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.14794"}},"official":{"repos":["google-research/google-research"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/mpnet-masked-and-permuted-pre-training-for","slug":"mpnet-masked-and-permuted-pre-training-for","title":"MPNet: Masked and Permuted Pre-training for Language Understanding","date":"2020-04-20","arxiv_id":"2004.09297","repositories_listed":7,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mpnet-masked-and-permuted-pre-training-for#ran","syntology_url":"https://syntology.ai/paper/2004.09297","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.09297"}},"official":{"repos":["microsoft/MPNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/flaubert-unsupervised-language-model-pre","slug":"flaubert-unsupervised-language-model-pre","title":"FlauBERT: Unsupervised Language Model Pre-training for French","date":"2019-12-11","arxiv_id":"1912.05372","repositories_listed":7,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/flaubert-unsupervised-language-model-pre#ran","syntology_url":"https://syntology.ai/paper/1912.05372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.05372"}},"official":{"repos":["getalp/Flaubert"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/plug-and-play-language-models-a-simple","slug":"plug-and-play-language-models-a-simple","title":"Plug and Play Language Models: A Simple Approach to Controlled Text Generation","date":"2019-12-04","arxiv_id":"1912.02164","repositories_listed":7,"syntology":{"n":22,"n_ran":17,"n_constructed":0,"n_ran_checked":15,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/plug-and-play-language-models-a-simple#ran","syntology_url":"https://syntology.ai/paper/1912.02164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.02164"}},"official":{"repos":["uber-research/PPLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/on-the-cross-lingual-transferability-of","slug":"on-the-cross-lingual-transferability-of","title":"On the Cross-lingual Transferability of Monolingual Representations","date":"2019-10-25","arxiv_id":"1910.11856","repositories_listed":7,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-the-cross-lingual-transferability-of#ran","syntology_url":"https://syntology.ai/paper/1910.11856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11856"}},"official":{"repos":["deepmind/xquad"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/uniter-learning-universal-image-text-1","slug":"uniter-learning-universal-image-text-1","title":"UNITER: UNiversal Image-TExt Representation Learning","date":"2019-09-25","arxiv_id":"1909.11740","repositories_listed":7,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uniter-learning-universal-image-text-1#ran","syntology_url":"https://syntology.ai/paper/1909.11740","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.11740"}},"official":{"repos":["ChenRocks/UNITER"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/large-memory-layers-with-product-keys","slug":"large-memory-layers-with-product-keys","title":"Large Memory Layers with Product Keys","date":"2019-07-10","arxiv_id":"1907.05242","repositories_listed":7,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-memory-layers-with-product-keys#ran","syntology_url":"https://syntology.ai/paper/1907.05242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.05242"}},"official":{"repos":["facebookresearch/XLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/190410509","slug":"190410509","title":"Generating Long Sequences with Sparse Transformers","date":"2019-04-23","arxiv_id":"1904.10509","repositories_listed":7,"syntology":{"n":6,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/190410509#ran","syntology_url":"https://syntology.ai/paper/1904.10509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.10509"}},"official":{"repos":["openai/sparse_attention"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/multi-task-deep-neural-networks-for-natural","slug":"multi-task-deep-neural-networks-for-natural","title":"Multi-Task Deep Neural Networks for Natural Language Understanding","date":"2019-01-31","arxiv_id":"1901.11504","repositories_listed":7,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/multi-task-deep-neural-networks-for-natural#ran","syntology_url":"https://syntology.ai/paper/1901.11504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.11504"}},"official":{"repos":["namisan/mt-dnn"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/ordered-neurons-integrating-tree-structures","slug":"ordered-neurons-integrating-tree-structures","title":"Ordered Neurons: Integrating Tree Structures into Recurrent Neural Networks","date":"2018-10-22","arxiv_id":"1810.09536","repositories_listed":7,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ordered-neurons-integrating-tree-structures#ran","syntology_url":"https://syntology.ai/paper/1810.09536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09536"}},"official":{"repos":["yikangshen/Ordered-Neurons"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/quasi-recurrent-neural-networks","slug":"quasi-recurrent-neural-networks","title":"Quasi-Recurrent Neural Networks","date":"2016-11-05","arxiv_id":"1611.01576","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quasi-recurrent-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1611.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01576"}},"official":null}},{"url":"/paper/qwen2-technical-report","slug":"qwen2-technical-report","title":"Qwen2 Technical Report","date":"2024-07-15","arxiv_id":"2407.10671","repositories_listed":6,"syntology":null},{"url":"/paper/chronos-learning-the-language-of-time-series","slug":"chronos-learning-the-language-of-time-series","title":"Chronos: Learning the Language of Time Series","date":"2024-03-12","arxiv_id":"2403.07815","repositories_listed":6,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":3,"n_violates":1,"n_no_contract":18,"n_pointer_only":5,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 3 honoured, 1 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/chronos-learning-the-language-of-time-series#ran","syntology_url":"https://syntology.ai/paper/2403.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07815"}},"official":{"repos":["SalesforceAIResearch/uni2ts","amazon-science/chronos-forecasting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/knowledge-graphs-meet-multi-modal-learning-a","slug":"knowledge-graphs-meet-multi-modal-learning-a","title":"Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey","date":"2024-02-08","arxiv_id":"2402.05391","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-graphs-meet-multi-modal-learning-a#ran","syntology_url":"https://syntology.ai/paper/2402.05391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05391"}},"official":{"repos":["zjukg/kg-mm-survey"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mixtral-of-experts#ran","syntology_url":"https://syntology.ai/paper/2401.04088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04088"}},"official":null}},{"url":"/paper/an-example-of-evolutionary-computation-large","slug":"an-example-of-evolutionary-computation-large","title":"Evolution of Heuristics: Towards Efficient Automatic Algorithm Design Using Large Language Model","date":"2024-01-04","arxiv_id":"2401.02051","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-example-of-evolutionary-computation-large#ran","syntology_url":"https://syntology.ai/paper/2401.02051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02051"}},"official":{"repos":["feiliu36/eoh"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gated-linear-attention-transformers-with","slug":"gated-linear-attention-transformers-with","title":"Gated Linear Attention Transformers with Hardware-Efficient Training","date":"2023-12-11","arxiv_id":"2312.06635","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gated-linear-attention-transformers-with#ran","syntology_url":"https://syntology.ai/paper/2312.06635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06635"}},"official":{"repos":["berlino/gated_linear_attention"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mistral-7b","slug":"mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mistral-7b#ran","syntology_url":"https://syntology.ai/paper/2310.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06825"}},"official":{"repos":["mistralai/mistral-src"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-streaming-language-models-with","slug":"efficient-streaming-language-models-with","title":"Efficient Streaming Language Models with Attention Sinks","date":"2023-09-29","arxiv_id":"2309.17453","repositories_listed":6,"syntology":{"n":11,"n_ran":8,"n_constructed":4,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-streaming-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.17453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17453"}},"official":{"repos":["intel/intel-extension-for-transformers","mit-han-lab/streaming-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deepspeed-ulysses-system-optimizations-for","slug":"deepspeed-ulysses-system-optimizations-for","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","date":"2023-09-25","arxiv_id":"2309.14509","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepspeed-ulysses-system-optimizations-for#ran","syntology_url":"https://syntology.ai/paper/2309.14509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14509"}},"official":null}},{"url":"/paper/flashattention-2-faster-attention-with-better","slug":"flashattention-2-faster-attention-with-better","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","date":"2023-07-17","arxiv_id":"2307.08691","repositories_listed":6,"syntology":null},{"url":"/paper/lost-in-the-middle-how-language-models-use","slug":"lost-in-the-middle-how-language-models-use","title":"Lost in the Middle: How Language Models Use Long Contexts","date":"2023-07-06","arxiv_id":"2307.03172","repositories_listed":6,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lost-in-the-middle-how-language-models-use#ran","syntology_url":"https://syntology.ai/paper/2307.03172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03172"}},"official":{"repos":["nelson-liu/lost-in-the-middle"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/tree-of-thoughts-deliberate-problem-solving-1","slug":"tree-of-thoughts-deliberate-problem-solving-1","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","date":"2023-05-17","arxiv_id":"2305.10601","repositories_listed":6,"syntology":{"n":24,"n_ran":19,"n_constructed":6,"n_ran_checked":18,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":1,"phrase":"19 ran (of which 6 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/tree-of-thoughts-deliberate-problem-solving-1#ran","syntology_url":"https://syntology.ai/paper/2305.10601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10601"}},"official":{"repos":["princeton-nlp/tree-of-thought-llm","ysymyth/tree-of-thought-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"url":"/paper/minigpt-4-enhancing-vision-language","slug":"minigpt-4-enhancing-vision-language","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","date":"2023-04-20","arxiv_id":"2304.10592","repositories_listed":6,"syntology":null},{"url":"/paper/a-survey-of-large-language-models","slug":"a-survey-of-large-language-models","title":"A Survey of Large Language Models","date":"2023-03-31","arxiv_id":"2303.18223","repositories_listed":6,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-survey-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2303.18223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.18223"}},"official":{"repos":["rucaibox/llmsurvey"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/massive-language-models-can-be-accurately","slug":"massive-language-models-can-be-accurately","title":"SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot","date":"2023-01-02","arxiv_id":"2301.00774","repositories_listed":6,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/massive-language-models-can-be-accurately#ran","syntology_url":"https://syntology.ai/paper/2301.00774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00774"}},"official":{"repos":["ist-daslab/sparsegpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/instructpix2pix-learning-to-follow-image","slug":"instructpix2pix-learning-to-follow-image","title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","date":"2022-11-17","arxiv_id":"2211.09800","repositories_listed":6,"syntology":{"n":20,"n_ran":17,"n_constructed":2,"n_ran_checked":12,"n_instrument":5,"n_unverified":3,"n_honours":2,"n_violates":3,"n_no_contract":7,"n_pointer_only":4,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 3 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/instructpix2pix-learning-to-follow-image#ran","syntology_url":"https://syntology.ai/paper/2211.09800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09800"}},"official":{"repos":["timothybrooks/instruct-pix2pix"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/audiolm-a-language-modeling-approach-to-audio","slug":"audiolm-a-language-modeling-approach-to-audio","title":"AudioLM: a Language Modeling Approach to Audio Generation","date":"2022-09-07","arxiv_id":"2209.03143","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 2 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/audiolm-a-language-modeling-approach-to-audio#ran","syntology_url":"https://syntology.ai/paper/2209.03143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03143"}},"official":null}},{"url":"/paper/pix2seq-a-language-modeling-framework-for","slug":"pix2seq-a-language-modeling-framework-for","title":"Pix2seq: A Language Modeling Framework for Object Detection","date":"2021-09-22","arxiv_id":"2109.10852","repositories_listed":6,"syntology":{"n":12,"n_ran":10,"n_constructed":3,"n_ran_checked":6,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":10,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pix2seq-a-language-modeling-framework-for#ran","syntology_url":"https://syntology.ai/paper/2109.10852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10852"}},"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bitfit-simple-parameter-efficient-fine-tuning","slug":"bitfit-simple-parameter-efficient-fine-tuning","title":"BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models","date":"2021-06-18","arxiv_id":"2106.10199","repositories_listed":6,"syntology":{"n":13,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/bitfit-simple-parameter-efficient-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2106.10199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10199"}},"official":{"repos":["benzakenelad/BitFit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/tsdae-using-transformer-based-sequential","slug":"tsdae-using-transformer-based-sequential","title":"TSDAE: Using Transformer-based Sequential Denoising Auto-Encoder for Unsupervised Sentence Embedding Learning","date":"2021-04-14","arxiv_id":"2104.06979","repositories_listed":6,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tsdae-using-transformer-based-sequential#ran","syntology_url":"https://syntology.ai/paper/2104.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06979"}},"official":{"repos":["kwang2049/pytorch-bertflow","kwang2049/useb","ukplab/pytorch-bertflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qa-gnn-reasoning-with-language-models-and","slug":"qa-gnn-reasoning-with-language-models-and","title":"QA-GNN: Reasoning with Language Models and Knowledge Graphs for Question Answering","date":"2021-04-13","arxiv_id":"2104.06378","repositories_listed":6,"syntology":{"n":25,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":17,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 17 unverified","sample_list":"/paper/qa-gnn-reasoning-with-language-models-and#ran","syntology_url":"https://syntology.ai/paper/2104.06378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06378"}},"official":{"repos":["michiyasunaga/qagnn","worksheets.codalab.org/worksheets/0xf215deb05edf44a2ac353c711f52a25f"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agnostic-bert-sentence-embedding","slug":"language-agnostic-bert-sentence-embedding","title":"Language-agnostic BERT Sentence Embedding","date":"2020-07-03","arxiv_id":"2007.01852","repositories_listed":6,"syntology":null},{"url":"/paper/contextnet-improving-convolutional-neural","slug":"contextnet-improving-convolutional-neural","title":"ContextNet: Improving Convolutional Neural Networks for Automatic Speech Recognition with Global Context","date":"2020-05-07","arxiv_id":"2005.03191","repositories_listed":6,"syntology":null},{"url":"/paper/revisiting-pre-trained-models-for-chinese","slug":"revisiting-pre-trained-models-for-chinese","title":"Revisiting Pre-Trained Models for Chinese Natural Language Processing","date":"2020-04-29","arxiv_id":"2004.13922","repositories_listed":6,"syntology":{"n":35,"n_ran":21,"n_constructed":1,"n_ran_checked":16,"n_instrument":5,"n_unverified":14,"n_honours":3,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"21 ran (of which 1 constructed an object rather than computing a result; 16 with no instrument failure: 3 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/revisiting-pre-trained-models-for-chinese#ran","syntology_url":"https://syntology.ai/paper/2004.13922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13922"}},"official":{"repos":["ymcui/MacBERT"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/realm-retrieval-augmented-language-model-pre","slug":"realm-retrieval-augmented-language-model-pre","title":"REALM: Retrieval-Augmented Language Model Pre-Training","date":"2020-02-10","arxiv_id":"2002.08909","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realm-retrieval-augmented-language-model-pre#ran","syntology_url":"https://syntology.ai/paper/2002.08909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08909"}},"official":{"repos":["google-research/language"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/exploiting-cloze-questions-for-few-shot-text","slug":"exploiting-cloze-questions-for-few-shot-text","title":"Exploiting Cloze Questions for Few Shot Text Classification and Natural Language Inference","date":"2020-01-21","arxiv_id":"2001.07676","repositories_listed":6,"syntology":null},{"url":"/paper/compressive-transformers-for-long-range-1","slug":"compressive-transformers-for-long-range-1","title":"Compressive Transformers for Long-Range Sequence Modelling","date":"2019-11-13","arxiv_id":"1911.05507","repositories_listed":6,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/compressive-transformers-for-long-range-1#ran","syntology_url":"https://syntology.ai/paper/1911.05507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.05507"}},"official":null}},{"url":"/paper/pseudolikelihood-reranking-with-masked","slug":"pseudolikelihood-reranking-with-masked","title":"Masked Language Model Scoring","date":"2019-10-31","arxiv_id":"1910.14659","repositories_listed":6,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pseudolikelihood-reranking-with-masked#ran","syntology_url":"https://syntology.ai/paper/1910.14659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.14659"}},"official":{"repos":["awslabs/mlm-scoring"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/enriching-pre-trained-language-model-with","slug":"enriching-pre-trained-language-model-with","title":"Enriching Pre-trained Language Model with Entity Information for Relation Classification","date":"2019-05-20","arxiv_id":"1905.08284","repositories_listed":6,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enriching-pre-trained-language-model-with#ran","syntology_url":"https://syntology.ai/paper/1905.08284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.08284"}},"official":null}},{"url":"/paper/fairseq-a-fast-extensible-toolkit-for","slug":"fairseq-a-fast-extensible-toolkit-for","title":"fairseq: A Fast, Extensible Toolkit for Sequence Modeling","date":"2019-04-01","arxiv_id":"1904.01038","repositories_listed":6,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/fairseq-a-fast-extensible-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/1904.01038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.01038"}},"official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/scibert-pretrained-contextualized-embeddings","slug":"scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","arxiv_id":"1903.10676","repositories_listed":6,"syntology":null},{"url":"/paper/federated-learning-for-mobile-keyboard","slug":"federated-learning-for-mobile-keyboard","title":"Federated Learning for Mobile Keyboard Prediction","date":"2018-11-08","arxiv_id":"1811.03604","repositories_listed":6,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/federated-learning-for-mobile-keyboard#ran","syntology_url":"https://syntology.ai/paper/1811.03604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.03604"}},"official":null}},{"url":"/paper/improving-generalization-performance-by","slug":"improving-generalization-performance-by","title":"Improving Generalization Performance by Switching from Adam to SGD","date":"2017-12-20","arxiv_id":"1712.07628","repositories_listed":6,"syntology":null},{"url":"/paper/deep-gradient-compression-reducing-the","slug":"deep-gradient-compression-reducing-the","title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","date":"2017-12-05","arxiv_id":"1712.01887","repositories_listed":6,"syntology":null},{"url":"/paper/advances-in-joint-ctc-attention-based-end-to","slug":"advances-in-joint-ctc-attention-based-end-to","title":"Advances in Joint CTC-Attention based End-to-End Speech Recognition with a Deep CNN Encoder and RNN-LM","date":"2017-06-08","arxiv_id":"1706.02737","repositories_listed":6,"syntology":null},{"url":"/paper/recurrent-highway-networks","slug":"recurrent-highway-networks","title":"Recurrent Highway Networks","date":"2016-07-12","arxiv_id":"1607.03474","repositories_listed":6,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/recurrent-highway-networks#ran","syntology_url":"https://syntology.ai/paper/1607.03474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1607.03474"}},"official":{"repos":["julian121266/RecurrentHighwayNetworks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"url":"/paper/sequence-to-sequence-learning-as-beam-search","slug":"sequence-to-sequence-learning-as-beam-search","title":"Sequence-to-Sequence Learning as Beam-Search Optimization","date":"2016-06-09","arxiv_id":"1606.02960","repositories_listed":6,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-to-sequence-learning-as-beam-search#ran","syntology_url":"https://syntology.ai/paper/1606.02960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02960"}},"official":{"repos":["harvardnlp/BSO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/recurrent-neural-network-grammars","slug":"recurrent-neural-network-grammars","title":"Recurrent Neural Network Grammars","date":"2016-02-25","arxiv_id":"1602.07776","repositories_listed":6,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recurrent-neural-network-grammars#ran","syntology_url":"https://syntology.ai/paper/1602.07776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.07776"}},"official":{"repos":["clab/rnng"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/a-simple-way-to-initialize-recurrent-networks","slug":"a-simple-way-to-initialize-recurrent-networks","title":"A Simple Way to Initialize Recurrent Networks of Rectified Linear Units","date":"2015-04-03","arxiv_id":"1504.00941","repositories_listed":6,"syntology":null},{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/scaling-and-evaluating-sparse-autoencoders","slug":"scaling-and-evaluating-sparse-autoencoders","title":"Scaling and evaluating sparse autoencoders","date":"2024-06-06","arxiv_id":"2406.04093","repositories_listed":5,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/scaling-and-evaluating-sparse-autoencoders#ran","syntology_url":"https://syntology.ai/paper/2406.04093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04093"}},"official":{"repos":["openai/sparse_autoencoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/transformers-are-ssms-generalized-models-and","slug":"transformers-are-ssms-generalized-models-and","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","date":"2024-05-31","arxiv_id":"2405.21060","repositories_listed":5,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/transformers-are-ssms-generalized-models-and#ran","syntology_url":"https://syntology.ai/paper/2405.21060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.21060"}},"official":{"repos":["state-spaces/mamba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deepseek-v2-a-strong-economical-and-efficient","slug":"deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","arxiv_id":"2405.04434","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepseek-v2-a-strong-economical-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.04434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04434"}},"official":{"repos":["deepseek-ai/deepseek-v2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/xlstm-extended-long-short-term-memory","slug":"xlstm-extended-long-short-term-memory","title":"xLSTM: Extended Long Short-Term Memory","date":"2024-05-07","arxiv_id":"2405.04517","repositories_listed":5,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/xlstm-extended-long-short-term-memory#ran","syntology_url":"https://syntology.ai/paper/2405.04517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04517"}},"official":{"repos":["nx-ai/xlstm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/leave-no-context-behind-efficient-infinite","slug":"leave-no-context-behind-efficient-infinite","title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","date":"2024-04-10","arxiv_id":"2404.07143","repositories_listed":5,"syntology":{"n":16,"n_ran":15,"n_constructed":4,"n_ran_checked":7,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":6,"phrase":"15 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leave-no-context-behind-efficient-infinite#ran","syntology_url":"https://syntology.ai/paper/2404.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07143"}},"official":null}},{"url":"/paper/llm-as-os-llmao-agents-as-apps-envisioning","slug":"llm-as-os-llmao-agents-as-apps-envisioning","title":"LLM as OS, Agents as Apps: Envisioning AIOS, Agents and the AIOS-Agent Ecosystem","date":"2023-12-06","arxiv_id":"2312.03815","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-as-os-llmao-agents-as-apps-envisioning#ran","syntology_url":"https://syntology.ai/paper/2312.03815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03815"}},"official":null}},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","slug":"point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/point-bind-point-llm-aligning-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2309.00615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00615"}},"official":{"repos":["ziyuguo99/point-bind_point-llm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/lima-less-is-more-for-alignment","slug":"lima-less-is-more-for-alignment","title":"LIMA: Less Is More for Alignment","date":"2023-05-18","arxiv_id":"2305.11206","repositories_listed":5,"syntology":null},{"url":"/paper/baize-an-open-source-chat-model-with","slug":"baize-an-open-source-chat-model-with","title":"Baize: An Open-Source Chat Model with Parameter-Efficient Tuning on Self-Chat Data","date":"2023-04-03","arxiv_id":"2304.01196","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/baize-an-open-source-chat-model-with#ran","syntology_url":"https://syntology.ai/paper/2304.01196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01196"}},"official":{"repos":["project-baize/baize","project-baize/baize-chatbot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/accelerating-large-language-model-decoding","slug":"accelerating-large-language-model-decoding","title":"Accelerating Large Language Model Decoding with Speculative Sampling","date":"2023-02-02","arxiv_id":"2302.01318","repositories_listed":5,"syntology":null},{"url":"/paper/muse-text-to-image-generation-via-masked","slug":"muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","arxiv_id":"2301.00704","repositories_listed":5,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":8,"n_instrument":11,"n_unverified":2,"n_honours":2,"n_violates":5,"n_no_contract":1,"n_pointer_only":11,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 5 violated, 1 with no contract checked; 11 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/muse-text-to-image-generation-via-masked#ran","syntology_url":"https://syntology.ai/paper/2301.00704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00704"}},"official":null}},{"url":"/paper/a-length-extrapolatable-transformer","slug":"a-length-extrapolatable-transformer","title":"A Length-Extrapolatable Transformer","date":"2022-12-20","arxiv_id":"2212.10554","repositories_listed":5,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-length-extrapolatable-transformer#ran","syntology_url":"https://syntology.ai/paper/2212.10554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10554"}},"official":{"repos":["microsoft/torchscale"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/transferring-textual-knowledge-for-visual","slug":"transferring-textual-knowledge-for-visual","title":"Revisiting Classifier: Transferring Vision-Language Models for Video Recognition","date":"2022-07-04","arxiv_id":"2207.01297","repositories_listed":5,"syntology":null},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/p-tuning-v2-prompt-tuning-can-be-comparable","slug":"p-tuning-v2-prompt-tuning-can-be-comparable","title":"P-Tuning v2: Prompt Tuning Can Be Comparable to Fine-tuning Universally Across Scales and Tasks","date":"2021-10-14","arxiv_id":"2110.07602","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/p-tuning-v2-prompt-tuning-can-be-comparable#ran","syntology_url":"https://syntology.ai/paper/2110.07602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07602"}},"official":{"repos":["thudm/p-tuning-v2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/from-two-to-one-a-new-scene-text-recognizer","slug":"from-two-to-one-a-new-scene-text-recognizer","title":"From Two to One: A New Scene Text Recognizer with Visual Language Modeling Network","date":"2021-08-22","arxiv_id":"2108.09661","repositories_listed":5,"syntology":null},{"url":"/paper/informer-transformer-likes-informed-attention","slug":"informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","arxiv_id":"2012.11747","repositories_listed":5,"syntology":null},{"url":"/paper/spelling-error-correction-with-soft-masked","slug":"spelling-error-correction-with-soft-masked","title":"Spelling Error Correction with Soft-Masked BERT","date":"2020-05-15","arxiv_id":"2005.07421","repositories_listed":5,"syntology":{"n":17,"n_ran":14,"n_constructed":2,"n_ran_checked":13,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"14 ran (of which 2 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spelling-error-correction-with-soft-masked#ran","syntology_url":"https://syntology.ai/paper/2005.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.07421"}},"official":null}},{"url":"/paper/document-level-representation-learning-using","slug":"document-level-representation-learning-using","title":"SPECTER: Document-level Representation Learning using Citation-informed Transformers","date":"2020-04-15","arxiv_id":"2004.07180","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/document-level-representation-learning-using#ran","syntology_url":"https://syntology.ai/paper/2004.07180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07180"}},"official":{"repos":["allenai/scidocs","allenai/specter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-probabilistic-formulation-of-unsupervised-1","slug":"a-probabilistic-formulation-of-unsupervised-1","title":"A Probabilistic Formulation of Unsupervised Text Style Transfer","date":"2020-02-10","arxiv_id":"2002.03912","repositories_listed":5,"syntology":{"n":25,"n_ran":18,"n_constructed":13,"n_ran_checked":11,"n_instrument":7,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"18 ran (of which 13 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/a-probabilistic-formulation-of-unsupervised-1#ran","syntology_url":"https://syntology.ai/paper/2002.03912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.03912"}},"official":{"repos":["cindyxinyiwang/deep-latent-sequence-model"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/single-headed-attention-rnn-stop-thinking","slug":"single-headed-attention-rnn-stop-thinking","title":"Single Headed Attention RNN: Stop Thinking With Your Head","date":"2019-11-26","arxiv_id":"1911.11423","repositories_listed":5,"syntology":null},{"url":"/paper/generalization-through-memorization-nearest","slug":"generalization-through-memorization-nearest","title":"Generalization through Memorization: Nearest Neighbor Language Models","date":"2019-11-01","arxiv_id":"1911.00172","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-through-memorization-nearest#ran","syntology_url":"https://syntology.ai/paper/1911.00172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.00172"}},"official":{"repos":["urvashik/knnlm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}}],"record_sha256":"944929c158cac144547cc9170ceda9d01d03b7d590a0317eaeb4aa5b212daf0c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}