{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/t5/papers/7","list_of":"/method/t5","method":"T5","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":8,"rows_per_page":100,"rows":[601,700],"of":708,"counts":{"archive_papers_tagged":708,"with_a_code_link":353,"where_syntology_ran_a_sample":99,"not_listed_spam_title":0,"listed":708,"listed_where_code_ran":99,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":85,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":85,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/t5","prev":"/method/t5/papers/6","next":"/method/t5/papers/8","papers":[{"paper":null,"slug":"sharpness-aware-minimization-improves","title":"Sharpness-Aware Minimization Improves Language Model Generalization","date":"2021-10-16","arxiv_id":"2110.08529","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-power-of-prompt-tuning-for-low-resource","title":"The Power of Prompt Tuning for Low-Resource Semantic Parsing","date":"2021-10-16","arxiv_id":"2110.08525","n_code_links":0,"syntology":null},{"paper":"/paper/lfpt5-a-unified-framework-for-lifelong-few-1","slug":"lfpt5-a-unified-framework-for-lifelong-few-1","title":"LFPT5: A Unified Framework for Lifelong Few-shot Language Learning Based on Prompt Tuning of T5","date":"2021-10-14","arxiv_id":"2110.07298","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["qcwthu/lifelong-fewshot-language-learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/speecht5-unified-modal-encoder-decoder-pre","slug":"speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","arxiv_id":"2110.07205","n_code_links":6,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/speecht5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"kg-fid-infusing-knowledge-graph-in-fusion-in-1","title":"KG-FiD: Infusing Knowledge Graph in Fusion-in-Decoder for Open-Domain Question Answering","date":"2021-10-08","arxiv_id":"2110.04330","n_code_links":0,"syntology":null},{"paper":"/paper/deepa2-a-modular-framework-for-deep-argument","slug":"deepa2-a-modular-framework-for-deep-argument","title":"DeepA2: A Modular Framework for Deep Argument Analysis with Pretrained Neural Text2Text Language Models","date":"2021-10-04","arxiv_id":"2110.01509","n_code_links":1,"syntology":null},{"paper":null,"slug":"perhaps-ptlms-should-go-to-school-a-task-to","title":"Perhaps PTLMs Should Go to School -- A Task to Assess Open Book and Closed Book QA","date":"2021-10-04","arxiv_id":"2110.01552","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-frequency-names-exhibit-bias-and","title":"Low Frequency Names Exhibit Bias and Overfitting in Contextualizing Language Models","date":"2021-10-01","arxiv_id":"2110.00672","n_code_links":0,"syntology":null},{"paper":null,"slug":"scale-efficiently-insights-from-pretraining","title":"Scale Efficiently: Insights from Pretraining and Finetuning Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"transslowdown-efficiency-attacks-on-neural","title":"TransSlowDown: Efficiency Attacks on Neural Machine Translation Systems","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/scale-efficiently-insights-from-pre-training","slug":"scale-efficiently-insights-from-pre-training","title":"Scale Efficiently: Insights from Pre-training and Fine-tuning Transformers","date":"2021-09-22","arxiv_id":"2109.10686","n_code_links":3,"syntology":{"ran":12,"of":12,"n_ran_checked":12,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 4 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"hierarchy-aware-t5-with-path-adaptive-mask","title":"Hierarchy-Aware T5 with Path-Adaptive Mask Mechanism for Hierarchical Text Classification","date":"2021-09-17","arxiv_id":"2109.08585","n_code_links":0,"syntology":null},{"paper":"/paper/primer-searching-for-efficient-transformers","slug":"primer-searching-for-efficient-transformers","title":"Primer: Searching for Efficient Transformers for Language Modeling","date":"2021-09-17","arxiv_id":"2109.08668","n_code_links":4,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/google-research"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"paper":null,"slug":"remixers-a-mixer-transformer-architecture","title":"Remixers: A Mixer-Transformer Architecture with Compositional Operators for Natural Language Understanding","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/language-models-are-few-shot-multilingual","slug":"language-models-are-few-shot-multilingual","title":"Language Models are Few-shot Multilingual Learners","date":"2021-09-16","arxiv_id":"2109.07684","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gentaiscool/few-shot-lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/attention-is-indeed-all-you-need-semantically","slug":"attention-is-indeed-all-you-need-semantically","title":"Attention Is Indeed All You Need: Semantically Attention-Guided Decoding for Data-to-Text NLG","date":"2021-09-15","arxiv_id":"2109.07043","n_code_links":1,"syntology":null},{"paper":null,"slug":"prefix-to-sql-text-to-sql-generation-from","title":"Prefix-to-SQL: Text-to-SQL Generation from Incomplete User Questions","date":"2021-09-15","arxiv_id":"2109.13066","n_code_links":0,"syntology":null},{"paper":"/paper/topic-transferable-table-question-answering","slug":"topic-transferable-table-question-answering","title":"Topic Transferable Table Question Answering","date":"2021-09-15","arxiv_id":"2109.07377","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-a-unified-sequence-to-sequence","slug":"exploring-a-unified-sequence-to-sequence","title":"Exploring a Unified Sequence-To-Sequence Transformer for Medical Product Safety Monitoring in Social Media","date":"2021-09-13","arxiv_id":"2109.05815","n_code_links":1,"syntology":null},{"paper":"/paper/investigating-numeracy-learning-ability-of-a","slug":"investigating-numeracy-learning-ability-of-a","title":"Investigating Numeracy Learning Ability of a Text-to-Text Transfer Model","date":"2021-09-10","arxiv_id":"2109.04672","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kuntalkumarpal/t5numeracy"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/picard-parsing-incrementally-for-constrained","slug":"picard-parsing-incrementally-for-constrained","title":"PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models","date":"2021-09-10","arxiv_id":"2109.05093","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ElementAI/picard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-dialogue-state-tracking-via-cross","slug":"zero-shot-dialogue-state-tracking-via-cross","title":"Zero-Shot Dialogue State Tracking via Cross-Task Transfer","date":"2021-09-10","arxiv_id":"2109.04655","n_code_links":1,"syntology":null},{"paper":null,"slug":"retrieve-caption-generate-visual-grounding","title":"Retrieve, Caption, Generate: Visual Grounding for Enhancing Commonsense in Text Generation Models","date":"2021-09-08","arxiv_id":"2109.03892","n_code_links":0,"syntology":null},{"paper":"/paper/general-purpose-question-answering-with-macaw","slug":"general-purpose-question-answering-with-macaw","title":"General-Purpose Question-Answering with Macaw","date":"2021-09-06","arxiv_id":"2109.02593","n_code_links":2,"syntology":null},{"paper":"/paper/codet5-identifier-aware-unified-pre-trained","slug":"codet5-identifier-aware-unified-pre-trained","title":"CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation","date":"2021-09-02","arxiv_id":"2109.00859","n_code_links":5,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["salesforce/codet5"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":"/paper/arat5-text-to-text-transformers-for-arabic","slug":"arat5-text-to-text-transformers-for-arabic","title":"AraT5: Text-to-Text Transformers for Arabic Language Generation","date":"2021-08-31","arxiv_id":"2109.12068","n_code_links":1,"syntology":null},{"paper":"/paper/sentence-t5-scalable-sentence-encoders-from","slug":"sentence-t5-scalable-sentence-encoders-from","title":"Sentence-T5: Scalable Sentence Encoders from Pre-trained Text-to-Text Models","date":"2021-08-19","arxiv_id":"2108.08877","n_code_links":2,"syntology":null},{"paper":"/paper/table-caption-generation-in-scholarly","slug":"table-caption-generation-in-scholarly","title":"Table Caption Generation in Scholarly Documents Leveraging Pre-trained Language Models","date":"2021-08-18","arxiv_id":"2108.08111","n_code_links":1,"syntology":null},{"paper":null,"slug":"enct5-fine-tuning-t5-encoder-for","title":"EncT5: Fine-tuning T5 Encoder for Discriminative Tasks","date":"2021-08-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sapphire-approaches-for-enhanced-concept-to","title":"SAPPHIRE: Approaches for Enhanced Concept-to-Text Generation","date":"2021-08-15","arxiv_id":"2108.06643","n_code_links":0,"syntology":null},{"paper":"/paper/how-optimal-is-greedy-decoding-for-extractive","slug":"how-optimal-is-greedy-decoding-for-extractive","title":"How Optimal is Greedy Decoding for Extractive Question Answering?","date":"2021-08-12","arxiv_id":"2108.05857","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ocastel/exact-extract"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-listwise-evidence-reasoning-with-t5","title":"Exploring Listwise Evidence Reasoning with T5 for Fact Verification","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"nmt5-is-parallel-data-still-relevant-for-pre-1","title":"nmT5 - Is parallel data still relevant for pre-training massively multilingual language models?","date":"2021-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/youngsheldon-at-semeval-2021-task-5-fine","slug":"youngsheldon-at-semeval-2021-task-5-fine","title":"YoungSheldon at SemEval-2021 Task 5: Fine-tuning Pre-trained Language Models for Toxic Spans Detection using Token classification Objective","date":"2021-08-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/emailsum-abstractive-email-thread","slug":"emailsum-abstractive-email-thread","title":"EmailSum: Abstractive Email Thread Summarization","date":"2021-07-30","arxiv_id":"2107.14691","n_code_links":1,"syntology":null},{"paper":"/paper/turning-tables-generating-examples-from-semi","slug":"turning-tables-generating-examples-from-semi","title":"Turning Tables: Generating Examples from Semi-structured Tables for Endowing Language Models with Reasoning Skills","date":"2021-07-15","arxiv_id":"2107.07261","n_code_links":1,"syntology":null},{"paper":"/paper/ernie-3-0-large-scale-knowledge-enhanced-pre","slug":"ernie-3-0-large-scale-knowledge-enhanced-pre","title":"ERNIE 3.0: Large-scale Knowledge Enhanced Pre-training for Language Understanding and Generation","date":"2021-07-05","arxiv_id":"2107.02137","n_code_links":2,"syntology":null},{"paper":null,"slug":"improving-factual-consistency-of-abstractive-1","title":"Improving Factual Consistency of Abstractive Summarization on Customer Feedback","date":"2021-06-30","arxiv_id":"2106.16188","n_code_links":0,"syntology":null},{"paper":"/paper/xl-sum-large-scale-multilingual-abstractive","slug":"xl-sum-large-scale-multilingual-abstractive","title":"XL-Sum: Large-Scale Multilingual Abstractive Summarization for 44 Languages","date":"2021-06-25","arxiv_id":"2106.13822","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["csebuetnlp/xl-sum"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"winner-team-mia-at-textvqa-challenge-2021","title":"Winner Team Mia at TextVQA Challenge 2021: Vision-and-Language Representation Learning with Pre-trained Sequence-to-Sequence Model","date":"2021-06-24","arxiv_id":"2106.15332","n_code_links":0,"syntology":null},{"paper":"/paper/cpm-2-large-scale-cost-effective-pre-trained","slug":"cpm-2-large-scale-cost-effective-pre-trained","title":"CPM-2: Large-scale Cost-effective Pre-trained Language Models","date":"2021-06-20","arxiv_id":"2106.10715","n_code_links":2,"syntology":null},{"paper":"/paper/jointgt-graph-text-joint-representation","slug":"jointgt-graph-text-joint-representation","title":"JointGT: Graph-Text Joint Representation Learning for Text Generation from Knowledge Graphs","date":"2021-06-19","arxiv_id":"2106.10502","n_code_links":1,"syntology":null},{"paper":"/paper/improving-paraphrase-detection-with-the","slug":"improving-paraphrase-detection-with-the","title":"Improving Paraphrase Detection with the Adversarial Paraphrasing Task","date":"2021-06-14","arxiv_id":"2106.07691","n_code_links":1,"syntology":null},{"paper":"/paper/memory-efficient-transformers-via-top-k","slug":"memory-efficient-transformers-via-top-k","title":"Memory-efficient Transformers via Top-$k$ Attention","date":"2021-06-13","arxiv_id":"2106.06899","n_code_links":2,"syntology":null},{"paper":"/paper/fastseq-make-sequence-generation-faster","slug":"fastseq-make-sequence-generation-faster","title":"FastSeq: Make Sequence Generation Faster","date":"2021-06-08","arxiv_id":"2106.04718","n_code_links":1,"syntology":null},{"paper":"/paper/timedial-temporal-commonsense-reasoning-in","slug":"timedial-temporal-commonsense-reasoning-in","title":"TIMEDIAL: Temporal Commonsense Reasoning in Dialog","date":"2021-06-08","arxiv_id":"2106.04571","n_code_links":1,"syntology":null},{"paper":null,"slug":"nmt5-is-parallel-data-still-relevant-for-pre","title":"nmT5 -- Is parallel data still relevant for pre-training massively multilingual language models?","date":"2021-06-03","arxiv_id":"2106.02171","n_code_links":0,"syntology":null},{"paper":"/paper/evidence-based-factual-error-correction","slug":"evidence-based-factual-error-correction","title":"Evidence-based Factual Error Correction","date":"2021-06-02","arxiv_id":"2106.01072","n_code_links":1,"syntology":null},{"paper":"/paper/implicit-representations-of-meaning-in-neural","slug":"implicit-representations-of-meaning-in-neural","title":"Implicit Representations of Meaning in Neural Language Models","date":"2021-06-01","arxiv_id":"2106.00737","n_code_links":1,"syntology":null},{"paper":"/paper/byt5-towards-a-token-free-future-with-pre","slug":"byt5-towards-a-token-free-future-with-pre","title":"ByT5: Towards a token-free future with pre-trained byte-to-byte models","date":"2021-05-28","arxiv_id":"2105.13626","n_code_links":5,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/byt5"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/scifive-a-text-to-text-transformer-model-for","slug":"scifive-a-text-to-text-transformer-model-for","title":"SciFive: a text-to-text transformer model for biomedical literature","date":"2021-05-28","arxiv_id":"2106.03598","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["justinphan3110/SciFive"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"attention-as-inference-via-fenchel-duality","title":"Attention as Inference via Fenchel Duality","date":"2021-05-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-text-to-text-transformers-for","title":"Exploring Text-to-Text Transformers for English to Hinglish Machine Translation with Synthetic Code-Mixing","date":"2021-05-18","arxiv_id":"2105.08807","n_code_links":0,"syntology":null},{"paper":"/paper/stage-wise-fine-tuning-for-graph-to-text","slug":"stage-wise-fine-tuning-for-graph-to-text","title":"Stage-wise Fine-tuning for Graph-to-Text Generation","date":"2021-05-17","arxiv_id":"2105.08021","n_code_links":1,"syntology":null},{"paper":null,"slug":"which-transformer-architecture-fits-my-data-a","title":"Which transformer architecture fits my data? A vocabulary bottleneck in self-attention","date":"2021-05-09","arxiv_id":"2105.03928","n_code_links":0,"syntology":null},{"paper":"/paper/mt6-multilingual-pretrained-text-to-text","slug":"mt6-multilingual-pretrained-text-to-text","title":"MT6: Multilingual Pretrained Text-to-Text Transformer with Translation Pairs","date":"2021-04-18","arxiv_id":"2104.08692","n_code_links":2,"syntology":null},{"paper":"/paper/the-power-of-scale-for-parameter-efficient","slug":"the-power-of-scale-for-parameter-efficient","title":"The Power of Scale for Parameter-Efficient Prompt Tuning","date":"2021-04-18","arxiv_id":"2104.08691","n_code_links":12,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google-research/prompt-tuning"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/decrypting-cryptic-crosswords-semantically","slug":"decrypting-cryptic-crosswords-semantically","title":"Decrypting Cryptic Crosswords: Semantically Complex Wordplay Puzzles as a Target for NLP","date":"2021-04-17","arxiv_id":"2104.08620","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-survey-of-recent-abstract-summarization","title":"A Survey of Recent Abstract Summarization Techniques","date":"2021-04-15","arxiv_id":"2105.00824","n_code_links":0,"syntology":null},{"paper":"/paper/explagraphs-an-explanation-graph-generation","slug":"explagraphs-an-explanation-graph-generation","title":"ExplaGraphs: An Explanation Graph Generation Task for Structured Commonsense Reasoning","date":"2021-04-15","arxiv_id":"2104.07644","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["swarnaHub/ExplaGraphs"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/nt5-training-t5-to-perform-numerical","slug":"nt5-training-t5-to-perform-numerical","title":"NT5?! Training T5 to Perform Numerical Reasoning","date":"2021-04-15","arxiv_id":"2104.07307","n_code_links":1,"syntology":null},{"paper":"/paper/nareor-the-narrative-reordering-problem","slug":"nareor-the-narrative-reordering-problem","title":"NAREOR: The Narrative Reordering Problem","date":"2021-04-14","arxiv_id":"2104.06669","n_code_links":1,"syntology":null},{"paper":null,"slug":"ki-bert-infusing-knowledge-context-for-better","title":"KI-BERT: Infusing Knowledge Context for Better Language and Domain Understanding","date":"2021-04-09","arxiv_id":"2104.08145","n_code_links":0,"syntology":null},{"paper":"/paper/codetrans-towards-cracking-the-language-of","slug":"codetrans-towards-cracking-the-language-of","title":"CodeTrans: Towards Cracking the Language of Silicon's Code Through Self-Supervised Deep Learning and High Performance Computing","date":"2021-04-06","arxiv_id":"2104.02443","n_code_links":1,"syntology":null},{"paper":"/paper/russian-paraphrasers-paraphrase-with","slug":"russian-paraphrasers-paraphrase-with","title":"Russian Paraphrasers: Paraphrase with Transformers","date":"2021-04-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":null,"slug":"automatic-graph-partitioning-for-very-large","title":"Automatic Graph Partitioning for Very Large-scale Deep Learning","date":"2021-03-30","arxiv_id":"2103.16063","n_code_links":0,"syntology":null},{"paper":"/paper/all-nlp-tasks-are-generation-tasks-a-general","slug":"all-nlp-tasks-are-generation-tasks-a-general","title":"GLM: General Language Model Pretraining with Autoregressive Blank Infilling","date":"2021-03-18","arxiv_id":"2103.10360","n_code_links":8,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/GLM"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/syntax-bert-improving-pre-trained","slug":"syntax-bert-improving-pre-trained","title":"Syntax-BERT: Improving Pre-trained Transformers with Syntax Trees","date":"2021-03-07","arxiv_id":"2103.04350","n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-task-transfer-learning-for-finding","title":"Multi-task transfer learning for finding actionable information from crisis-related messages on social media","date":"2021-02-26","arxiv_id":"2102.13395","n_code_links":0,"syntology":null},{"paper":"/paper/pada-a-prompt-based-autoregressive-approach","slug":"pada-a-prompt-based-autoregressive-approach","title":"PADA: Example-based Prompt Learning for on-the-fly Adaptation to Unseen Domains","date":"2021-02-24","arxiv_id":"2102.12206","n_code_links":1,"syntology":null},{"paper":"/paper/quiz-style-question-generation-for-news","slug":"quiz-style-question-generation-for-news","title":"Quiz-Style Question Generation for News Stories","date":"2021-02-18","arxiv_id":"2102.09094","n_code_links":2,"syntology":null},{"paper":null,"slug":"transformer-based-models-for-question","title":"Transformer-Based Models for Question Answering on COVID19","date":"2021-01-16","arxiv_id":"2101.11432","n_code_links":0,"syntology":null},{"paper":"/paper/improving-sequence-to-sequence-pre-training","slug":"improving-sequence-to-sequence-pre-training","title":"Improving Sequence-to-Sequence Pre-training via Sequence Span Rewriting","date":"2021-01-02","arxiv_id":"2101.00416","n_code_links":1,"syntology":null},{"paper":null,"slug":"pre-training-text-to-text-transformers-to","title":"Pre-training Text-to-Text Transformers to Write and Reason with Concepts","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/earlybert-efficient-bert-training-via-early-1","slug":"earlybert-efficient-bert-training-via-early-1","title":"EarlyBERT: Efficient BERT Training via Early-bird Lottery Tickets","date":"2020-12-31","arxiv_id":"2101.00063","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["VITA-Group/EarlyBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/factual-error-correction-of-claims","slug":"factual-error-correction-of-claims","title":"Evidence-based Factual Error Correction","date":"2020-12-31","arxiv_id":"2012.15788","n_code_links":3,"syntology":null},{"paper":"/paper/leveraging-parsbert-and-pretrained-mt5-for","slug":"leveraging-parsbert-and-pretrained-mt5-for","title":"Leveraging ParsBERT and Pretrained mT5 for Persian Abstractive Text Summarization","date":"2020-12-21","arxiv_id":"2012.11204","n_code_links":1,"syntology":null},{"paper":"/paper/how-can-we-know-when-language-models-know","slug":"how-can-we-know-when-language-models-know","title":"How Can We Know When Language Models Know? On the Calibration of Language Models for Question Answering","date":"2020-12-02","arxiv_id":"2012.00955","n_code_links":1,"syntology":null},{"paper":"/paper/flight-of-the-pegasus-comparing-transformers","slug":"flight-of-the-pegasus-comparing-transformers","title":"Flight of the PEGASUS? Comparing Transformers on Few-shot and Zero-shot Multi-document Abstractive Summarization","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"label-representations-in-modeling","title":"Label Representations in Modeling Classification as Text Generation","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-task-descriptions","slug":"learning-from-task-descriptions","title":"Learning from Task Descriptions","date":"2020-11-16","arxiv_id":"2011.08115","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/seqgensql-a-robust-sequence-generation-model","slug":"seqgensql-a-robust-sequence-generation-model","title":"SeqGenSQL -- A Robust Sequence Generation Model for Structured Query Language","date":"2020-11-07","arxiv_id":"2011.03836","n_code_links":2,"syntology":null},{"paper":"/paper/abnirml-analyzing-the-behavior-of-neural-ir","slug":"abnirml-analyzing-the-behavior-of-neural-ir","title":"ABNIRML: Analyzing the Behavior of Neural IR Models","date":"2020-11-02","arxiv_id":"2011.00696","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allenai/abnirml"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-zero-shot-conditional-summarization","slug":"towards-zero-shot-conditional-summarization","title":"Towards Zero-Shot Conditional Summarization with Adaptive Multi-Task Fine-Tuning","date":"2020-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/measuring-association-between-labels-and-free","slug":"measuring-association-between-labels-and-free","title":"Measuring Association Between Labels and Free-Text Rationales","date":"2020-10-24","arxiv_id":"2010.12762","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/label_rationale_association"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"unsupervised-paraphrase-generation-via","title":"Unsupervised Paraphrasing with Pretrained Language Models","date":"2020-10-24","arxiv_id":"2010.12885","n_code_links":0,"syntology":null},{"paper":"/paper/mt5-a-massively-multilingual-pre-trained-text","slug":"mt5-a-massively-multilingual-pre-trained-text","title":"mT5: A massively multilingual pre-trained text-to-text transformer","date":"2020-10-22","arxiv_id":"2010.11934","n_code_links":8,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["google-research/multilingual-t5"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"scientific-claim-verification-with-vert5erini","title":"Scientific Claim Verification with VERT5ERINI","date":"2020-10-22","arxiv_id":"2010.11930","n_code_links":0,"syntology":null},{"paper":"/paper/parameter-norm-growth-during-training-of","slug":"parameter-norm-growth-during-training-of","title":"Effects of Parameter Norm Growth During Transformer Training: Inductive Bias from Gradient Descent","date":"2020-10-19","arxiv_id":"2010.09697","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":1,"n_instrument":5,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["viking-sudo-rm/norm-growth"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"chatbot-interaction-with-artificial","title":"Chatbot Interaction with Artificial Intelligence: Human Data Augmentation with T5 and Language Transformer Ensemble for Text Classification","date":"2020-10-12","arxiv_id":"2010.05990","n_code_links":0,"syntology":null},{"paper":"/paper/textsettr-label-free-text-style-extraction-1","slug":"textsettr-label-free-text-style-extraction-1","title":"TextSETTR: Few-Shot Text Style Extraction and Tunable Targeted Restyling","date":"2020-10-08","arxiv_id":"2010.03802","n_code_links":1,"syntology":null},{"paper":"/paper/converting-the-point-of-view-of-messages","slug":"converting-the-point-of-view-of-messages","title":"Converting the Point of View of Messages Spoken to Virtual Assistants","date":"2020-10-06","arxiv_id":"2010.02600","n_code_links":2,"syntology":null},{"paper":"/paper/mintl-minimalist-transfer-learning-for-task","slug":"mintl-minimalist-transfer-learning-for-task","title":"MinTL: Minimalist Transfer Learning for Task-Oriented Dialogue Systems","date":"2020-09-25","arxiv_id":"2009.12005","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zlinao/MinTL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ucd-cs-at-w-nut-2020-shared-task-3-a-text-to","slug":"ucd-cs-at-w-nut-2020-shared-task-3-a-text-to","title":"UCD-CS at W-NUT 2020 Shared Task-3: A Text to Text Approach for COVID-19 Event Extraction on Social Media","date":"2020-09-21","arxiv_id":"2009.10047","n_code_links":1,"syntology":null},{"paper":"/paper/lite-training-strategies-for-portuguese","slug":"lite-training-strategies-for-portuguese","title":"Lite Training Strategies for Portuguese-English and English-Portuguese Translation","date":"2020-08-20","arxiv_id":"2008.08769","n_code_links":1,"syntology":null},{"paper":"/paper/parade-passage-representation-aggregation-for","slug":"parade-passage-representation-aggregation-for","title":"PARADE: Passage Representation Aggregation for Document Reranking","date":"2020-08-20","arxiv_id":"2008.09093","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["canjiali/PARADE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":"/paper/ptt5-pretraining-and-validating-the-t5-model","slug":"ptt5-pretraining-and-validating-the-t5-model","title":"PTT5: Pretraining and validating the T5 model on Brazilian Portuguese data","date":"2020-08-20","arxiv_id":"2008.09144","n_code_links":3,"syntology":null},{"paper":"/paper/investigating-pretrained-language-models-for","slug":"investigating-pretrained-language-models-for","title":"Investigating Pretrained Language Models for Graph-to-Text Generation","date":"2020-07-16","arxiv_id":"2007.08426","n_code_links":3,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["UKPLab/plms-graph2text"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hypergrid-efficient-multi-task-transformers","title":"HyperGrid: Efficient Multi-Task Transformers with Grid-wise Decomposable Hyper Projections","date":"2020-07-12","arxiv_id":"2007.05891","n_code_links":0,"syntology":null},{"paper":null,"slug":"normalizador-neural-de-datas-e-enderecos","title":"Normalizador Neural de Datas e Endereços","date":"2020-06-27","arxiv_id":"2007.04300","n_code_links":0,"syntology":null}],"record_sha256":"0c84d76d501feea1f90a76b6aae77f0b4fe27bd11d178c841d12b650f989befc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}