{"url":"/task/memorization","name":"Memorization","slug":"memorization","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1088,"papers_with_code":438,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/memorization-on-big-bench-hindu-knowledge","slug":"memorization-on-big-bench-hindu-knowledge","dataset":"BIG-bench (Hindu Knowledge)","dataset_url":"/dataset/big-bench","rows_in_archive":3,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"PaLM-540B (few-shot, k=5)","paper_title":"PaLM: Scaling Language Modeling with Pathways","paper_url":"/paper/palm-scaling-language-modeling-with-pathways-1","paper_date":"2022-04-05","arxiv_id":"2204.02311","code_links":[{"title":"lucidrains/CoCa-pytorch","url":"https://github.com/lucidrains/CoCa-pytorch"},{"title":"lucidrains/PaLM-pytorch","url":"https://github.com/lucidrains/PaLM-pytorch"},{"title":"google/paxml","url":"https://github.com/google/paxml"},{"title":"foundation-model-stack/fms-fsdp","url":"https://github.com/foundation-model-stack/fms-fsdp"},{"title":"lucidrains/PaLM-jax","url":"https://github.com/lucidrains/PaLM-jax"},{"title":"chrisociepa/allamo","url":"https://github.com/chrisociepa/allamo"},{"title":"conceptofmind/PaLM-flax","url":"https://github.com/conceptofmind/PaLM-flax"}],"syntology":{"n":37,"n_ran":30,"n_unverified":7,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/big-bench","name":"BIG-bench","full_name":"Beyond the Imitation Game Benchmark","num_papers_in_archive":349},{"url":"/dataset/popqa","name":"PopQA","full_name":"","num_papers_in_archive":91},{"url":"/dataset/ds-1000","name":"DS-1000","full_name":"","num_papers_in_archive":72},{"url":"/dataset/lm-email-address-leakage","name":"LM Email Address Leakage","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":438,"tagged_in_all":1088,"items":[{"url":"/paper/mixup-beyond-empirical-risk-minimization","title":"mixup: Beyond Empirical Risk Minimization","date":"2017-10-25","arxiv_id":"1710.09412","repositories_listed":71,"syntology":{"n":47,"n_ran":30,"n_unverified":17,"n_pointer_only":15}},{"url":"/paper/wide-deep-learning-for-recommender-systems","title":"Wide & Deep Learning for Recommender Systems","date":"2016-06-24","arxiv_id":"1606.07792","repositories_listed":39,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":5}},{"url":"/paper/neural-machine-translation-in-linear-time","title":"Neural Machine Translation in Linear Time","date":"2016-10-31","arxiv_id":"1610.10099","repositories_listed":11,"syntology":null},{"url":"/paper/palm-scaling-language-modeling-with-pathways-1","title":"PaLM: Scaling Language Modeling with Pathways","date":"2022-04-05","arxiv_id":"2204.02311","repositories_listed":7,"syntology":{"n":37,"n_ran":30,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/beyond-the-imitation-game-quantifying-and","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","date":"2022-06-09","arxiv_id":"2206.04615","repositories_listed":6,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/grokking-generalization-beyond-overfitting-on","title":"Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","date":"2022-01-06","arxiv_id":"2201.02177","repositories_listed":6,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/generalization-through-memorization-nearest","title":"Generalization through Memorization: Nearest Neighbor Language Models","date":"2019-11-01","arxiv_id":"1911.00172","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/co-teaching-robust-training-of-deep-neural","title":"Co-teaching: Robust Training of Deep Neural Networks with Extremely Noisy Labels","date":"2018-04-18","arxiv_id":"1804.06872","repositories_listed":5,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/exposing-flaws-of-generative-model-evaluation-1","title":"Exposing flaws of generative model evaluation metrics and their unfair treatment of diffusion models","date":"2023-06-07","arxiv_id":"2306.04675","repositories_listed":4,"syntology":{"n":63,"n_ran":30,"n_unverified":33,"n_pointer_only":1}},{"url":"/paper/pythia-a-suite-for-analyzing-large-language","title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","date":"2023-04-03","arxiv_id":"2304.01373","repositories_listed":4,"syntology":null},{"url":"/paper/learning-with-noisy-labels-revisited-a-study-1","title":"Learning with Noisy Labels Revisited: A Study Using Real-World Human Annotations","date":"2021-10-22","arxiv_id":"2110.12088","repositories_listed":4,"syntology":null},{"url":"/paper/learning-explanations-that-are-hard-to-vary","title":"Learning explanations that are hard to vary","date":"2020-09-01","arxiv_id":"2009.00329","repositories_listed":4,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/das3h-modeling-student-learning-and","title":"DAS3H: Modeling Student Learning and Forgetting for Optimally Scheduling Distributed Practice of Skills","date":"2019-05-14","arxiv_id":"1905.06873","repositories_listed":4,"syntology":{"n":22,"n_ran":5,"n_unverified":17,"n_pointer_only":9}},{"url":"/paper/r1-searcher-incentivizing-the-dynamic","title":"R1-Searcher++: Incentivizing the Dynamic Knowledge Acquisition of LLMs via Reinforcement Learning","date":"2025-05-22","arxiv_id":"2505.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":1,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/limo-less-is-more-for-reasoning","title":"LIMO: Less is More for Reasoning","date":"2025-02-05","arxiv_id":"2502.03387","repositories_listed":3,"syntology":null},{"url":"/paper/muse-machine-unlearning-six-way-evaluation","title":"MUSE: Machine Unlearning Six-Way Evaluation for Language Models","date":"2024-07-08","arxiv_id":"2407.06460","repositories_listed":3,"syntology":null},{"url":"/paper/evaluating-inexact-unlearning-requires","title":"Towards Adversarial Evaluations for Inexact Machine Unlearning","date":"2022-01-17","arxiv_id":"2201.06640","repositories_listed":3,"syntology":null},{"url":"/paper/scaling-language-models-methods-analysis-1","title":"Scaling Language Models: Methods, Analysis & Insights from Training Gopher","date":"2021-12-08","arxiv_id":"2112.11446","repositories_listed":3,"syntology":null},{"url":"/paper/consensual-collaborative-training-and","title":"Consensual Collaborative Training And Knowledge Distillation Based Facial Expression Recognition Under Noisy Annotations","date":"2021-07-10","arxiv_id":"2107.04746","repositories_listed":3,"syntology":null},{"url":"/paper/searching-to-exploit-memorization-effect-in","title":"Searching to Exploit Memorization Effect in Learning from Corrupted Labels","date":"2019-11-06","arxiv_id":"1911.02377","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/how-does-disagreement-help-generalization","title":"How does Disagreement Help Generalization against Label Corruption?","date":"2019-01-14","arxiv_id":"1901.04215","repositories_listed":3,"syntology":null},{"url":"/paper/associative-long-short-term-memory","title":"Associative Long Short-Term Memory","date":"2016-02-09","arxiv_id":"1602.03032","repositories_listed":3,"syntology":null},{"url":"/paper/llm-srbench-a-new-benchmark-for-scientific","title":"LLM-SRBench: A New Benchmark for Scientific Equation Discovery with Large Language Models","date":"2025-04-14","arxiv_id":"2504.10415","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/videochat-flash-hierarchical-compression-for","title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","date":"2024-12-31","arxiv_id":"2501.00574","repositories_listed":2,"syntology":null},{"url":"/paper/dash-warm-starting-neural-network-training-in","title":"DASH: Warm-Starting Neural Network Training in Stationary Settings without Loss of Plasticity","date":"2024-10-30","arxiv_id":"2410.23495","repositories_listed":2,"syntology":{"n":14,"n_ran":3,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/how-do-large-language-models-acquire-factual","title":"How Do Large Language Models Acquire Factual Knowledge During Pretraining?","date":"2024-06-17","arxiv_id":"2406.11813","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/disentangled-continual-learning-separating","title":"Infinite dSprites for Disentangled Continual Learning: Separating Memory Edits from Generalization","date":"2023-12-27","arxiv_id":"2312.16731","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/negative-pre-aware-for-noisy-cross-modal","title":"Negative Pre-aware for Noisy Cross-modal Matching","date":"2023-12-10","arxiv_id":"2312.05777","repositories_listed":2,"syntology":null},{"url":"/paper/data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/practical-membership-inference-attacks","title":"Practical Membership Inference Attacks against Fine-tuned Large Language Models via Self-prompt Calibration","date":"2023-11-10","arxiv_id":"2311.06062","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":6}}],"syntology_records":18,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}