{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/2","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":142,"rows_per_page":100,"rows":[101,200],"of":14182,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling","prev":"/task/language-modeling","next":"/task/language-modeling/papers/3","papers":[{"url":"/paper/ordered-neurons-integrating-tree-structures","slug":"ordered-neurons-integrating-tree-structures","title":"Ordered Neurons: Integrating Tree Structures into Recurrent Neural Networks","date":"2018-10-22","arxiv_id":"1810.09536","repositories_listed":7,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ordered-neurons-integrating-tree-structures#ran","syntology_url":"https://syntology.ai/paper/1810.09536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.09536"}},"official":{"repos":["yikangshen/Ordered-Neurons"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/quasi-recurrent-neural-networks","slug":"quasi-recurrent-neural-networks","title":"Quasi-Recurrent Neural Networks","date":"2016-11-05","arxiv_id":"1611.01576","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quasi-recurrent-neural-networks#ran","syntology_url":"https://syntology.ai/paper/1611.01576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.01576"}},"official":null}},{"url":"/paper/chronos-learning-the-language-of-time-series","slug":"chronos-learning-the-language-of-time-series","title":"Chronos: Learning the Language of Time Series","date":"2024-03-12","arxiv_id":"2403.07815","repositories_listed":6,"syntology":{"n":28,"n_ran":23,"n_constructed":0,"n_ran_checked":22,"n_instrument":1,"n_unverified":5,"n_honours":3,"n_violates":1,"n_no_contract":18,"n_pointer_only":5,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 22 with no instrument failure: 3 honoured, 1 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/chronos-learning-the-language-of-time-series#ran","syntology_url":"https://syntology.ai/paper/2403.07815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07815"}},"official":{"repos":["SalesforceAIResearch/uni2ts","amazon-science/chronos-forecasting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/knowledge-graphs-meet-multi-modal-learning-a","slug":"knowledge-graphs-meet-multi-modal-learning-a","title":"Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey","date":"2024-02-08","arxiv_id":"2402.05391","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-graphs-meet-multi-modal-learning-a#ran","syntology_url":"https://syntology.ai/paper/2402.05391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05391"}},"official":{"repos":["zjukg/kg-mm-survey"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mixtral-of-experts#ran","syntology_url":"https://syntology.ai/paper/2401.04088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04088"}},"official":null}},{"url":"/paper/an-example-of-evolutionary-computation-large","slug":"an-example-of-evolutionary-computation-large","title":"Evolution of Heuristics: Towards Efficient Automatic Algorithm Design Using Large Language Model","date":"2024-01-04","arxiv_id":"2401.02051","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-example-of-evolutionary-computation-large#ran","syntology_url":"https://syntology.ai/paper/2401.02051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02051"}},"official":{"repos":["feiliu36/eoh"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gated-linear-attention-transformers-with","slug":"gated-linear-attention-transformers-with","title":"Gated Linear Attention Transformers with Hardware-Efficient Training","date":"2023-12-11","arxiv_id":"2312.06635","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gated-linear-attention-transformers-with#ran","syntology_url":"https://syntology.ai/paper/2312.06635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06635"}},"official":{"repos":["berlino/gated_linear_attention"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mistral-7b","slug":"mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mistral-7b#ran","syntology_url":"https://syntology.ai/paper/2310.06825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06825"}},"official":{"repos":["mistralai/mistral-src"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/efficient-streaming-language-models-with","slug":"efficient-streaming-language-models-with","title":"Efficient Streaming Language Models with Attention Sinks","date":"2023-09-29","arxiv_id":"2309.17453","repositories_listed":6,"syntology":{"n":11,"n_ran":8,"n_constructed":4,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/efficient-streaming-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.17453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17453"}},"official":{"repos":["intel/intel-extension-for-transformers","mit-han-lab/streaming-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/flashattention-2-faster-attention-with-better","slug":"flashattention-2-faster-attention-with-better","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","date":"2023-07-17","arxiv_id":"2307.08691","repositories_listed":6,"syntology":null},{"url":"/paper/a-survey-of-large-language-models","slug":"a-survey-of-large-language-models","title":"A Survey of Large Language Models","date":"2023-03-31","arxiv_id":"2303.18223","repositories_listed":6,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/a-survey-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2303.18223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.18223"}},"official":{"repos":["rucaibox/llmsurvey"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/instructpix2pix-learning-to-follow-image","slug":"instructpix2pix-learning-to-follow-image","title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","date":"2022-11-17","arxiv_id":"2211.09800","repositories_listed":6,"syntology":{"n":20,"n_ran":17,"n_constructed":2,"n_ran_checked":12,"n_instrument":5,"n_unverified":3,"n_honours":2,"n_violates":3,"n_no_contract":7,"n_pointer_only":4,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 12 with no instrument failure: 2 honoured, 3 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/instructpix2pix-learning-to-follow-image#ran","syntology_url":"https://syntology.ai/paper/2211.09800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09800"}},"official":{"repos":["timothybrooks/instruct-pix2pix"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/audiolm-a-language-modeling-approach-to-audio","slug":"audiolm-a-language-modeling-approach-to-audio","title":"AudioLM: a Language Modeling Approach to Audio Generation","date":"2022-09-07","arxiv_id":"2209.03143","repositories_listed":6,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 2 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/audiolm-a-language-modeling-approach-to-audio#ran","syntology_url":"https://syntology.ai/paper/2209.03143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.03143"}},"official":null}},{"url":"/paper/pix2seq-a-language-modeling-framework-for","slug":"pix2seq-a-language-modeling-framework-for","title":"Pix2seq: A Language Modeling Framework for Object Detection","date":"2021-09-22","arxiv_id":"2109.10852","repositories_listed":6,"syntology":{"n":12,"n_ran":10,"n_constructed":3,"n_ran_checked":6,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":10,"phrase":"10 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pix2seq-a-language-modeling-framework-for#ran","syntology_url":"https://syntology.ai/paper/2109.10852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.10852"}},"official":{"repos":["google-research/pix2seq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bitfit-simple-parameter-efficient-fine-tuning","slug":"bitfit-simple-parameter-efficient-fine-tuning","title":"BitFit: Simple Parameter-efficient Fine-tuning for Transformer-based Masked Language-models","date":"2021-06-18","arxiv_id":"2106.10199","repositories_listed":6,"syntology":{"n":13,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/bitfit-simple-parameter-efficient-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2106.10199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.10199"}},"official":{"repos":["benzakenelad/BitFit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/tsdae-using-transformer-based-sequential","slug":"tsdae-using-transformer-based-sequential","title":"TSDAE: Using Transformer-based Sequential Denoising Auto-Encoder for Unsupervised Sentence Embedding Learning","date":"2021-04-14","arxiv_id":"2104.06979","repositories_listed":6,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tsdae-using-transformer-based-sequential#ran","syntology_url":"https://syntology.ai/paper/2104.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.06979"}},"official":{"repos":["kwang2049/pytorch-bertflow","kwang2049/useb","ukplab/pytorch-bertflow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agnostic-bert-sentence-embedding","slug":"language-agnostic-bert-sentence-embedding","title":"Language-agnostic BERT Sentence Embedding","date":"2020-07-03","arxiv_id":"2007.01852","repositories_listed":6,"syntology":null},{"url":"/paper/contextnet-improving-convolutional-neural","slug":"contextnet-improving-convolutional-neural","title":"ContextNet: Improving Convolutional Neural Networks for Automatic Speech Recognition with Global Context","date":"2020-05-07","arxiv_id":"2005.03191","repositories_listed":6,"syntology":null},{"url":"/paper/revisiting-pre-trained-models-for-chinese","slug":"revisiting-pre-trained-models-for-chinese","title":"Revisiting Pre-Trained Models for Chinese Natural Language Processing","date":"2020-04-29","arxiv_id":"2004.13922","repositories_listed":6,"syntology":{"n":35,"n_ran":21,"n_constructed":1,"n_ran_checked":16,"n_instrument":5,"n_unverified":14,"n_honours":3,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"21 ran (of which 1 constructed an object rather than computing a result; 16 with no instrument failure: 3 honoured, 0 violated, 13 with no contract checked; 5 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/revisiting-pre-trained-models-for-chinese#ran","syntology_url":"https://syntology.ai/paper/2004.13922","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.13922"}},"official":{"repos":["ymcui/MacBERT"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/realm-retrieval-augmented-language-model-pre","slug":"realm-retrieval-augmented-language-model-pre","title":"REALM: Retrieval-Augmented Language Model Pre-Training","date":"2020-02-10","arxiv_id":"2002.08909","repositories_listed":6,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realm-retrieval-augmented-language-model-pre#ran","syntology_url":"https://syntology.ai/paper/2002.08909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.08909"}},"official":{"repos":["google-research/language"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/exploiting-cloze-questions-for-few-shot-text","slug":"exploiting-cloze-questions-for-few-shot-text","title":"Exploiting Cloze Questions for Few Shot Text Classification and Natural Language Inference","date":"2020-01-21","arxiv_id":"2001.07676","repositories_listed":6,"syntology":null},{"url":"/paper/pseudolikelihood-reranking-with-masked","slug":"pseudolikelihood-reranking-with-masked","title":"Masked Language Model Scoring","date":"2019-10-31","arxiv_id":"1910.14659","repositories_listed":6,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pseudolikelihood-reranking-with-masked#ran","syntology_url":"https://syntology.ai/paper/1910.14659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.14659"}},"official":{"repos":["awslabs/mlm-scoring"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/enriching-pre-trained-language-model-with","slug":"enriching-pre-trained-language-model-with","title":"Enriching Pre-trained Language Model with Entity Information for Relation Classification","date":"2019-05-20","arxiv_id":"1905.08284","repositories_listed":6,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enriching-pre-trained-language-model-with#ran","syntology_url":"https://syntology.ai/paper/1905.08284","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.08284"}},"official":null}},{"url":"/paper/fairseq-a-fast-extensible-toolkit-for","slug":"fairseq-a-fast-extensible-toolkit-for","title":"fairseq: A Fast, Extensible Toolkit for Sequence Modeling","date":"2019-04-01","arxiv_id":"1904.01038","repositories_listed":6,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/fairseq-a-fast-extensible-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/1904.01038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.01038"}},"official":{"repos":["pytorch/fairseq"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/scibert-pretrained-contextualized-embeddings","slug":"scibert-pretrained-contextualized-embeddings","title":"SciBERT: A Pretrained Language Model for Scientific Text","date":"2019-03-26","arxiv_id":"1903.10676","repositories_listed":6,"syntology":null},{"url":"/paper/passage-re-ranking-with-bert","slug":"passage-re-ranking-with-bert","title":"Passage Re-ranking with BERT","date":"2019-01-13","arxiv_id":"1901.04085","repositories_listed":6,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/passage-re-ranking-with-bert#ran","syntology_url":"https://syntology.ai/paper/1901.04085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.04085"}},"official":{"repos":["nyu-dl/dl4marco-bert"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/federated-learning-for-mobile-keyboard","slug":"federated-learning-for-mobile-keyboard","title":"Federated Learning for Mobile Keyboard Prediction","date":"2018-11-08","arxiv_id":"1811.03604","repositories_listed":6,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/federated-learning-for-mobile-keyboard#ran","syntology_url":"https://syntology.ai/paper/1811.03604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.03604"}},"official":null}},{"url":"/paper/improving-generalization-performance-by","slug":"improving-generalization-performance-by","title":"Improving Generalization Performance by Switching from Adam to SGD","date":"2017-12-20","arxiv_id":"1712.07628","repositories_listed":6,"syntology":null},{"url":"/paper/deep-gradient-compression-reducing-the","slug":"deep-gradient-compression-reducing-the","title":"Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training","date":"2017-12-05","arxiv_id":"1712.01887","repositories_listed":6,"syntology":null},{"url":"/paper/advances-in-joint-ctc-attention-based-end-to","slug":"advances-in-joint-ctc-attention-based-end-to","title":"Advances in Joint CTC-Attention based End-to-End Speech Recognition with a Deep CNN Encoder and RNN-LM","date":"2017-06-08","arxiv_id":"1706.02737","repositories_listed":6,"syntology":null},{"url":"/paper/recurrent-highway-networks","slug":"recurrent-highway-networks","title":"Recurrent Highway Networks","date":"2016-07-12","arxiv_id":"1607.03474","repositories_listed":6,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/recurrent-highway-networks#ran","syntology_url":"https://syntology.ai/paper/1607.03474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1607.03474"}},"official":{"repos":["julian121266/RecurrentHighwayNetworks"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"url":"/paper/sequence-to-sequence-learning-as-beam-search","slug":"sequence-to-sequence-learning-as-beam-search","title":"Sequence-to-Sequence Learning as Beam-Search Optimization","date":"2016-06-09","arxiv_id":"1606.02960","repositories_listed":6,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequence-to-sequence-learning-as-beam-search#ran","syntology_url":"https://syntology.ai/paper/1606.02960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02960"}},"official":{"repos":["harvardnlp/BSO"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/recurrent-neural-network-grammars","slug":"recurrent-neural-network-grammars","title":"Recurrent Neural Network Grammars","date":"2016-02-25","arxiv_id":"1602.07776","repositories_listed":6,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recurrent-neural-network-grammars#ran","syntology_url":"https://syntology.ai/paper/1602.07776","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1602.07776"}},"official":{"repos":["clab/rnng"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/a-simple-way-to-initialize-recurrent-networks","slug":"a-simple-way-to-initialize-recurrent-networks","title":"A Simple Way to Initialize Recurrent Networks of Rectified Linear Units","date":"2015-04-03","arxiv_id":"1504.00941","repositories_listed":6,"syntology":null},{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/transformers-are-ssms-generalized-models-and","slug":"transformers-are-ssms-generalized-models-and","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","date":"2024-05-31","arxiv_id":"2405.21060","repositories_listed":5,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/transformers-are-ssms-generalized-models-and#ran","syntology_url":"https://syntology.ai/paper/2405.21060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.21060"}},"official":{"repos":["state-spaces/mamba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deepseek-v2-a-strong-economical-and-efficient","slug":"deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","arxiv_id":"2405.04434","repositories_listed":5,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepseek-v2-a-strong-economical-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.04434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04434"}},"official":{"repos":["deepseek-ai/deepseek-v2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/xlstm-extended-long-short-term-memory","slug":"xlstm-extended-long-short-term-memory","title":"xLSTM: Extended Long Short-Term Memory","date":"2024-05-07","arxiv_id":"2405.04517","repositories_listed":5,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/xlstm-extended-long-short-term-memory#ran","syntology_url":"https://syntology.ai/paper/2405.04517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04517"}},"official":{"repos":["nx-ai/xlstm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/leave-no-context-behind-efficient-infinite","slug":"leave-no-context-behind-efficient-infinite","title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","date":"2024-04-10","arxiv_id":"2404.07143","repositories_listed":5,"syntology":{"n":16,"n_ran":15,"n_constructed":4,"n_ran_checked":7,"n_instrument":8,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":6,"phrase":"15 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 8 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leave-no-context-behind-efficient-infinite#ran","syntology_url":"https://syntology.ai/paper/2404.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07143"}},"official":null}},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","slug":"point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":15,"n_constructed":0,"n_ran_checked":11,"n_instrument":4,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/point-bind-point-llm-aligning-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2309.00615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.00615"}},"official":{"repos":["ziyuguo99/point-bind_point-llm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/baize-an-open-source-chat-model-with","slug":"baize-an-open-source-chat-model-with","title":"Baize: An Open-Source Chat Model with Parameter-Efficient Tuning on Self-Chat Data","date":"2023-04-03","arxiv_id":"2304.01196","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/baize-an-open-source-chat-model-with#ran","syntology_url":"https://syntology.ai/paper/2304.01196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.01196"}},"official":{"repos":["project-baize/baize","project-baize/baize-chatbot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/accelerating-large-language-model-decoding","slug":"accelerating-large-language-model-decoding","title":"Accelerating Large Language Model Decoding with Speculative Sampling","date":"2023-02-02","arxiv_id":"2302.01318","repositories_listed":5,"syntology":null},{"url":"/paper/a-length-extrapolatable-transformer","slug":"a-length-extrapolatable-transformer","title":"A Length-Extrapolatable Transformer","date":"2022-12-20","arxiv_id":"2212.10554","repositories_listed":5,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-length-extrapolatable-transformer#ran","syntology_url":"https://syntology.ai/paper/2212.10554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10554"}},"official":{"repos":["microsoft/torchscale"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/p-tuning-v2-prompt-tuning-can-be-comparable","slug":"p-tuning-v2-prompt-tuning-can-be-comparable","title":"P-Tuning v2: Prompt Tuning Can Be Comparable to Fine-tuning Universally Across Scales and Tasks","date":"2021-10-14","arxiv_id":"2110.07602","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/p-tuning-v2-prompt-tuning-can-be-comparable#ran","syntology_url":"https://syntology.ai/paper/2110.07602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.07602"}},"official":{"repos":["thudm/p-tuning-v2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/from-two-to-one-a-new-scene-text-recognizer","slug":"from-two-to-one-a-new-scene-text-recognizer","title":"From Two to One: A New Scene Text Recognizer with Visual Language Modeling Network","date":"2021-08-22","arxiv_id":"2108.09661","repositories_listed":5,"syntology":null},{"url":"/paper/informer-transformer-likes-informed-attention","slug":"informer-transformer-likes-informed-attention","title":"RealFormer: Transformer Likes Residual Attention","date":"2020-12-21","arxiv_id":"2012.11747","repositories_listed":5,"syntology":null},{"url":"/paper/spelling-error-correction-with-soft-masked","slug":"spelling-error-correction-with-soft-masked","title":"Spelling Error Correction with Soft-Masked BERT","date":"2020-05-15","arxiv_id":"2005.07421","repositories_listed":5,"syntology":{"n":17,"n_ran":14,"n_constructed":2,"n_ran_checked":13,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"14 ran (of which 2 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/spelling-error-correction-with-soft-masked#ran","syntology_url":"https://syntology.ai/paper/2005.07421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.07421"}},"official":null}},{"url":"/paper/document-level-representation-learning-using","slug":"document-level-representation-learning-using","title":"SPECTER: Document-level Representation Learning using Citation-informed Transformers","date":"2020-04-15","arxiv_id":"2004.07180","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/document-level-representation-learning-using#ran","syntology_url":"https://syntology.ai/paper/2004.07180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.07180"}},"official":{"repos":["allenai/scidocs","allenai/specter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/single-headed-attention-rnn-stop-thinking","slug":"single-headed-attention-rnn-stop-thinking","title":"Single Headed Attention RNN: Stop Thinking With Your Head","date":"2019-11-26","arxiv_id":"1911.11423","repositories_listed":5,"syntology":null},{"url":"/paper/generalization-through-memorization-nearest","slug":"generalization-through-memorization-nearest","title":"Generalization through Memorization: Nearest Neighbor Language Models","date":"2019-11-01","arxiv_id":"1911.00172","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-through-memorization-nearest#ran","syntology_url":"https://syntology.ai/paper/1911.00172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.00172"}},"official":{"repos":["urvashik/knnlm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/stabilizing-transformers-for-reinforcement-1","slug":"stabilizing-transformers-for-reinforcement-1","title":"Stabilizing Transformers for Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.06764","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stabilizing-transformers-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/1910.06764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.06764"}},"official":null}},{"url":"/paper/reducing-transformer-depth-on-demand-with-1","slug":"reducing-transformer-depth-on-demand-with-1","title":"Reducing Transformer Depth on Demand with Structured Dropout","date":"2019-09-25","arxiv_id":"1909.11556","repositories_listed":5,"syntology":null},{"url":"/paper/conditional-bert-contextual-augmentation","slug":"conditional-bert-contextual-augmentation","title":"Conditional BERT Contextual Augmentation","date":"2018-12-17","arxiv_id":"1812.06705","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-bert-contextual-augmentation#ran","syntology_url":"https://syntology.ai/paper/1812.06705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.06705"}},"official":null}},{"url":"/paper/neural-abstractive-text-summarization-with","slug":"neural-abstractive-text-summarization-with","title":"Neural Abstractive Text Summarization with Sequence-to-Sequence Models","date":"2018-12-05","arxiv_id":"1812.02303","repositories_listed":5,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-abstractive-text-summarization-with#ran","syntology_url":"https://syntology.ai/paper/1812.02303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.02303"}},"official":{"repos":["tshi04/NATS"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/targeted-syntactic-evaluation-of-language","slug":"targeted-syntactic-evaluation-of-language","title":"Targeted Syntactic Evaluation of Language Models","date":"2018-08-27","arxiv_id":"1808.09031","repositories_listed":5,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/targeted-syntactic-evaluation-of-language#ran","syntology_url":"https://syntology.ai/paper/1808.09031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.09031"}},"official":{"repos":["BeckyMarvin/LM_syneval"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/neural-architecture-optimization","slug":"neural-architecture-optimization","title":"Neural Architecture Optimization","date":"2018-08-22","arxiv_id":"1808.07233","repositories_listed":5,"syntology":null},{"url":"/paper/online-spatial-concept-and-lexical","slug":"online-spatial-concept-and-lexical","title":"Online Spatial Concept and Lexical Acquisition with Simultaneous Localization and Mapping","date":"2017-04-15","arxiv_id":"1704.04664","repositories_listed":5,"syntology":null},{"url":"/paper/structured-sequence-modeling-with-graph","slug":"structured-sequence-modeling-with-graph","title":"Structured Sequence Modeling with Graph Convolutional Recurrent Networks","date":"2016-12-22","arxiv_id":"1612.07659","repositories_listed":5,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/structured-sequence-modeling-with-graph#ran","syntology_url":"https://syntology.ai/paper/1612.07659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.07659"}},"official":{"repos":["youngjoo-epfl/gconvRNN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-python-code-suggestion-with-a-sparse","slug":"learning-python-code-suggestion-with-a-sparse","title":"Learning Python Code Suggestion with a Sparse Pointer Network","date":"2016-11-24","arxiv_id":"1611.08307","repositories_listed":5,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-python-code-suggestion-with-a-sparse#ran","syntology_url":"https://syntology.ai/paper/1611.08307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1611.08307"}},"official":{"repos":["uclmr/pycodesuggest"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/assessing-the-ability-of-lstms-to-learn","slug":"assessing-the-ability-of-lstms-to-learn","title":"Assessing the Ability of LSTMs to Learn Syntax-Sensitive Dependencies","date":"2016-11-04","arxiv_id":"1611.01368","repositories_listed":5,"syntology":null},{"url":"/paper/tying-word-vectors-and-word-classifiers-a","slug":"tying-word-vectors-and-word-classifiers-a","title":"Tying Word Vectors and Word Classifiers: A Loss Framework for Language Modeling","date":"2016-11-04","arxiv_id":"1611.01462","repositories_listed":5,"syntology":null},{"url":"/paper/gated-word-character-recurrent-language-model","slug":"gated-word-character-recurrent-language-model","title":"Gated Word-Character Recurrent Language Model","date":"2016-06-06","arxiv_id":"1606.01700","repositories_listed":5,"syntology":null},{"url":"/paper/learning-longer-memory-in-recurrent-neural","slug":"learning-longer-memory-in-recurrent-neural","title":"Learning Longer Memory in Recurrent Neural Networks","date":"2014-12-24","arxiv_id":"1412.7753","repositories_listed":5,"syntology":null},{"url":"/paper/first-pass-large-vocabulary-continuous-speech","slug":"first-pass-large-vocabulary-continuous-speech","title":"First-Pass Large Vocabulary Continuous Speech Recognition using Bi-Directional Recurrent DNNs","date":"2014-08-12","arxiv_id":"1408.2873","repositories_listed":5,"syntology":null},{"url":"/paper/word2vec-explained-deriving-mikolov-et-als","slug":"word2vec-explained-deriving-mikolov-et-als","title":"word2vec Explained: deriving Mikolov et al.'s negative-sampling word-embedding method","date":"2014-02-15","arxiv_id":"1402.3722","repositories_listed":5,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/word2vec-explained-deriving-mikolov-et-als#ran","syntology_url":"https://syntology.ai/paper/1402.3722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1402.3722"}},"official":null}},{"url":"/paper/mamut-a-novel-framework-for-modifying","slug":"mamut-a-novel-framework-for-modifying","title":"MAMUT: A Novel Framework for Modifying Mathematical Formulas for the Generation of Specialized Datasets for Language Model Training","date":"2025-02-28","arxiv_id":"2502.20855","repositories_listed":4,"syntology":null},{"url":"/paper/deepseek-v3-technical-report","slug":"deepseek-v3-technical-report","title":"DeepSeek-V3 Technical Report","date":"2024-12-27","arxiv_id":"2412.19437","repositories_listed":4,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deepseek-v3-technical-report#ran","syntology_url":"https://syntology.ai/paper/2412.19437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19437"}},"official":{"repos":["deepseek-ai/deepseek-v3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gated-delta-networks-improving-mamba2-with","slug":"gated-delta-networks-improving-mamba2-with","title":"Gated Delta Networks: Improving Mamba2 with Delta Rule","date":"2024-12-09","arxiv_id":"2412.06464","repositories_listed":4,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":7,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gated-delta-networks-improving-mamba2-with#ran","syntology_url":"https://syntology.ai/paper/2412.06464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06464"}},"official":{"repos":["NVlabs/GatedDeltaNet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/the-ademamix-optimizer-better-faster-older","slug":"the-ademamix-optimizer-better-faster-older","title":"The AdEMAMix Optimizer: Better, Faster, Older","date":"2024-09-05","arxiv_id":"2409.03137","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-ademamix-optimizer-better-faster-older#ran","syntology_url":"https://syntology.ai/paper/2409.03137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03137"}},"official":{"repos":["apple/ml-ademamix"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-synthetic-data-creation-with","slug":"scaling-synthetic-data-creation-with","title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","date":"2024-06-28","arxiv_id":"2406.20094","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-synthetic-data-creation-with#ran","syntology_url":"https://syntology.ai/paper/2406.20094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20094"}},"official":{"repos":["tencent-ailab/persona-hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/ranking-manipulation-for-conversational","slug":"ranking-manipulation-for-conversational","title":"Ranking Manipulation for Conversational Search Engines","date":"2024-06-05","arxiv_id":"2406.03589","repositories_listed":4,"syntology":{"n":20,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":20,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/ranking-manipulation-for-conversational#ran","syntology_url":"https://syntology.ai/paper/2406.03589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03589"}},"official":{"repos":["spfrommer/ragdoll-data-pipeline","spfrommer/ranking_manipulation","spfrommer/ranking_manipulation_data_pipeline","spfrommer/cse-ranking-manipulation"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/xmodel-vlm-a-simple-baseline-for-multimodal","slug":"xmodel-vlm-a-simple-baseline-for-multimodal","title":"Xmodel-VLM: A Simple Baseline for Multimodal Vision Language Model","date":"2024-05-15","arxiv_id":"2405.09215","repositories_listed":4,"syntology":null},{"url":"/paper/openelm-an-efficient-language-model-family","slug":"openelm-an-efficient-language-model-family","title":"OpenELM: An Efficient Language Model Family with Open Training and Inference Framework","date":"2024-04-22","arxiv_id":"2404.14619","repositories_listed":4,"syntology":null},{"url":"/paper/hgrn2-gated-linear-rnns-with-state-expansion","slug":"hgrn2-gated-linear-rnns-with-state-expansion","title":"HGRN2: Gated Linear RNNs with State Expansion","date":"2024-04-11","arxiv_id":"2404.07904","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hgrn2-gated-linear-rnns-with-state-expansion#ran","syntology_url":"https://syntology.ai/paper/2404.07904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07904"}},"official":{"repos":["sustcsonglin/flash-linear-attention","opennlplab/hgrn2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-time-series-classification-with","slug":"advancing-time-series-classification-with","title":"Advancing Time Series Classification with Multimodal Language Modeling","date":"2024-03-19","arxiv_id":"2403.12371","repositories_listed":4,"syntology":null},{"url":"/paper/tower-an-open-multilingual-large-language","slug":"tower-an-open-multilingual-large-language","title":"Tower: An Open Multilingual Large Language Model for Translation-Related Tasks","date":"2024-02-27","arxiv_id":"2402.17733","repositories_listed":4,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tower-an-open-multilingual-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.17733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17733"}},"official":{"repos":["deep-spin/tower-eval","epfllm/megatron-llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/repetition-improves-language-model-embeddings","slug":"repetition-improves-language-model-embeddings","title":"Repetition Improves Language Model Embeddings","date":"2024-02-23","arxiv_id":"2402.15449","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/repetition-improves-language-model-embeddings#ran","syntology_url":"https://syntology.ai/paper/2402.15449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15449"}},"official":{"repos":["jakespringer/echo-embeddings"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/algorithm-evolution-using-large-language","slug":"algorithm-evolution-using-large-language","title":"Algorithm Evolution Using Large Language Model","date":"2023-11-26","arxiv_id":"2311.15249","repositories_listed":4,"syntology":null},{"url":"/paper/chat-univi-unified-visual-representation","slug":"chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","arxiv_id":"2311.08046","repositories_listed":4,"syntology":null},{"url":"/paper/cogvlm-visual-expert-for-pretrained-language","slug":"cogvlm-visual-expert-for-pretrained-language","title":"CogVLM: Visual Expert for Pretrained Language Models","date":"2023-11-06","arxiv_id":"2311.03079","repositories_listed":4,"syntology":null},{"url":"/paper/nlp-evaluation-in-trouble-on-the-need-to","slug":"nlp-evaluation-in-trouble-on-the-need-to","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","date":"2023-10-27","arxiv_id":"2310.18018","repositories_listed":4,"syntology":null},{"url":"/paper/discrete-diffusion-language-modeling-by","slug":"discrete-diffusion-language-modeling-by","title":"Discrete Diffusion Modeling by Estimating the Ratios of the Data Distribution","date":"2023-10-25","arxiv_id":"2310.16834","repositories_listed":4,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":14,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":13,"n_pointer_only":15,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discrete-diffusion-language-modeling-by#ran","syntology_url":"https://syntology.ai/paper/2310.16834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16834"}},"official":{"repos":["louaaron/score-entropy-discrete-diffusion"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llemma-an-open-language-model-for-mathematics","slug":"llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","arxiv_id":"2310.10631","repositories_listed":4,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llemma-an-open-language-model-for-mathematics#ran","syntology_url":"https://syntology.ai/paper/2310.10631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10631"}},"official":{"repos":["EleutherAI/math-lm","eleutherai/gpt-neox","wellecks/llmstep"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/neftune-noisy-embeddings-improve-instruction","slug":"neftune-noisy-embeddings-improve-instruction","title":"NEFTune: Noisy Embeddings Improve Instruction Finetuning","date":"2023-10-09","arxiv_id":"2310.05914","repositories_listed":4,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neftune-noisy-embeddings-improve-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.05914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05914"}},"official":{"repos":["neelsjain/neftune"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/expertqa-expert-curated-questions-and","slug":"expertqa-expert-curated-questions-and","title":"ExpertQA: Expert-Curated Questions and Attributed Answers","date":"2023-09-14","arxiv_id":"2309.07852","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/expertqa-expert-curated-questions-and#ran","syntology_url":"https://syntology.ai/paper/2309.07852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07852"}},"official":{"repos":["chaitanyamalaviya/expertqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mme-a-comprehensive-evaluation-benchmark-for","slug":"mme-a-comprehensive-evaluation-benchmark-for","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13394","repositories_listed":4,"syntology":null},{"url":"/paper/video-llama-an-instruction-tuned-audio-visual","slug":"video-llama-an-instruction-tuned-audio-visual","title":"Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding","date":"2023-06-05","arxiv_id":"2306.02858","repositories_listed":4,"syntology":{"n":25,"n_ran":18,"n_constructed":7,"n_ran_checked":8,"n_instrument":10,"n_unverified":7,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":9,"phrase":"18 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 10 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/video-llama-an-instruction-tuned-audio-visual#ran","syntology_url":"https://syntology.ai/paper/2306.02858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02858"}},"official":{"repos":["damo-nlp-sg/video-llama"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gqa-training-generalized-multi-query","slug":"gqa-training-generalized-multi-query","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","date":"2023-05-22","arxiv_id":"2305.13245","repositories_listed":4,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gqa-training-generalized-multi-query#ran","syntology_url":"https://syntology.ai/paper/2305.13245","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13245"}},"official":null}},{"url":"/paper/how-to-index-item-ids-for-recommendation","slug":"how-to-index-item-ids-for-recommendation","title":"How to Index Item IDs for Recommendation Foundation Models","date":"2023-05-11","arxiv_id":"2305.06569","repositories_listed":4,"syntology":null},{"url":"/paper/deeptextmark-deep-learning-based-text","slug":"deeptextmark-deep-learning-based-text","title":"DeepTextMark: A Deep Learning-Driven Text Watermarking Approach for Identifying Large Language Model Generated Text","date":"2023-05-09","arxiv_id":"2305.05773","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deeptextmark-deep-learning-based-text#ran","syntology_url":"https://syntology.ai/paper/2305.05773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05773"}},"official":null}},{"url":"/paper/lamp-when-large-language-models-meet","slug":"lamp-when-large-language-models-meet","title":"LaMP: When Large Language Models Meet Personalization","date":"2023-04-22","arxiv_id":"2304.11406","repositories_listed":4,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lamp-when-large-language-models-meet#ran","syntology_url":"https://syntology.ai/paper/2304.11406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.11406"}},"official":null}},{"url":"/paper/winclip-zero-few-shot-anomaly-classification","slug":"winclip-zero-few-shot-anomaly-classification","title":"WinCLIP: Zero-/Few-Shot Anomaly Classification and Segmentation","date":"2023-03-26","arxiv_id":"2303.14814","repositories_listed":4,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":5,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/winclip-zero-few-shot-anomaly-classification#ran","syntology_url":"https://syntology.ai/paper/2303.14814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14814"}},"official":null}},{"url":"/paper/foundation-transformers","slug":"foundation-transformers","title":"Foundation Transformers","date":"2022-10-12","arxiv_id":"2210.06423","repositories_listed":4,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foundation-transformers#ran","syntology_url":"https://syntology.ai/paper/2210.06423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06423"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","slug":"pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pix2struct-screenshot-parsing-as-pretraining#ran","syntology_url":"https://syntology.ai/paper/2210.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03347"}},"official":{"repos":["google-research/pix2struct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/layoutlmv3-pre-training-for-document-ai-with","slug":"layoutlmv3-pre-training-for-document-ai-with","title":"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking","date":"2022-04-18","arxiv_id":"2204.08387","repositories_listed":4,"syntology":null},{"url":"/paper/memorizing-transformers-1","slug":"memorizing-transformers-1","title":"Memorizing Transformers","date":"2022-03-16","arxiv_id":"2203.08913","repositories_listed":4,"syntology":{"n":34,"n_ran":23,"n_constructed":6,"n_ran_checked":18,"n_instrument":5,"n_unverified":11,"n_honours":3,"n_violates":1,"n_no_contract":14,"n_pointer_only":2,"phrase":"23 ran (of which 6 constructed an object rather than computing a result; 18 with no instrument failure: 3 honoured, 1 violated, 14 with no contract checked; 5 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/memorizing-transformers-1#ran","syntology_url":"https://syntology.ai/paper/2203.08913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08913"}},"official":{"repos":["lucidrains/memorizing-transformers-pytorch"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clipcap-clip-prefix-for-image-captioning","slug":"clipcap-clip-prefix-for-image-captioning","title":"ClipCap: CLIP Prefix for Image Captioning","date":"2021-11-18","arxiv_id":"2111.09734","repositories_listed":4,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clipcap-clip-prefix-for-image-captioning#ran","syntology_url":"https://syntology.ai/paper/2111.09734","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.09734"}},"official":{"repos":["rmokady/clip_prefix_caption"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"c1329d92457dd8b9aaf0ace5b8395473e151c4487e62825d3c44e46437f05748","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}