{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/memorization/papers/8","list_of":"/task/memorization","task":"Memorization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":11,"rows_per_page":100,"rows":[701,800],"of":1088,"counts":{"archive_papers_tagged":1088,"with_a_code_link":438,"where_syntology_ran_a_sample":177,"not_listed_spam_title":0,"listed":1088,"listed_where_code_ran":177,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":148,"every_run_a_failure_of_syntologys_instrument":29,"listed_with_a_run_with_no_instrument_failure":148,"listed_every_run_a_failure_of_syntologys_instrument":29,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/memorization","prev":"/task/memorization/papers/7","next":"/task/memorization/papers/9","papers":[{"url":null,"slug":"learnable-privacy-neurons-localization-in","title":"Learnable Privacy Neurons Localization in Language Models","date":"2024-05-16","arxiv_id":"2405.10989","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-loss-decay-based-robust-oriented","title":"Dynamic Loss Decay based Robust Oriented Object Detection on Remote Sensing Images with Noisy Labels","date":"2024-05-15","arxiv_id":"2405.09024","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-holographic-reduced","title":"Generalized Holographic Reduced Representations","date":"2024-05-15","arxiv_id":"2405.09689","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-scaling-laws-understanding-transformer","title":"Beyond Scaling Laws: Understanding Transformer Performance with Associative Memory","date":"2024-05-14","arxiv_id":"2405.08707","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-tokens-for-language-modeling","title":"Thinking Tokens for Language Modeling","date":"2024-05-14","arxiv_id":"2405.08644","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-each-textual-sequence-its-own-improving","title":"To Each (Textual Sequence) Its Own: Improving Memorized-Data Unlearning in Large Language Models","date":"2024-05-06","arxiv_id":"2405.03097","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-prompts-to-elicit-memorization-in","title":"Exploring prompts to elicit memorization in masked language model-based named entity recognition","date":"2024-05-05","arxiv_id":"2405.03004","repositories_listed":0,"syntology":null},{"url":null,"slug":"mothman-at-semeval-2024-task-9-an-iterative","title":"Mothman at SemEval-2024 Task 9: An Iterative System for Chain-of-Thought Prompt Optimization","date":"2024-05-03","arxiv_id":"2405.02517","repositories_listed":0,"syntology":null},{"url":null,"slug":"report-on-the-aapm-grand-challenge-on-deep","title":"Report on the AAPM Grand Challenge on deep generative modeling for learning medical image statistics","date":"2024-05-03","arxiv_id":"2405.01822","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-memorization-of-domain-specific","title":"Quantifying Memorization and Detecting Training Data of Pre-trained Language Models using Japanese Newspaper","date":"2024-04-26","arxiv_id":"2404.17143","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-llm-memorization-through-the-lens","title":"Rethinking LLM Memorization through the Lens of Adversarial Compression","date":"2024-04-23","arxiv_id":"2404.15146","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-model-watermarking-defending-against","title":"Reliable Model Watermarking: Defending Against Theft without Compromising on Evasion","date":"2024-04-21","arxiv_id":"2404.13518","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-positivity-of-the-neural-tangent-kernel","title":"The Positivity of the Neural Tangent Kernel","date":"2024-04-19","arxiv_id":"2404.12928","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-assessment-of-prompts-used-in-code","title":"The Fault in our Stars: Quality Assessment of Code Generation Benchmarks","date":"2024-04-15","arxiv_id":"2404.10155","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-knowledge-and-reasoning-emulating-expert","title":"AI Knowledge and Reasoning: Emulating Expert Creativity in Scientific Research","date":"2024-04-05","arxiv_id":"2404.04436","repositories_listed":0,"syntology":null},{"url":null,"slug":"gp-molformer-a-foundation-model-for-molecular","title":"GP-MoLFormer: A Foundation Model For Molecular Generation","date":"2024-04-04","arxiv_id":"2405.04912","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-better-generalization-in-open-domain","title":"Towards Better Generalization in Open-Domain Question Answering by Mitigating Context Memorization","date":"2024-04-02","arxiv_id":"2404.01652","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-transformer-learn-with-varying-depth","title":"What Can Transformer Learn with Varying Depth? Case Studies on Sequence Learning Tasks","date":"2024-04-02","arxiv_id":"2404.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"sok-a-review-of-differentially-private-linear","title":"SoK: A Review of Differentially Private Linear Models For High-Dimensional Data","date":"2024-04-01","arxiv_id":"2404.01141","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-memorization-free-diffusion-models","title":"Towards Memorization-Free Diffusion Models","date":"2024-04-01","arxiv_id":"2404.00922","repositories_listed":0,"syntology":null},{"url":"/paper/language-models-learn-rare-phenomena-from","slug":"language-models-learn-rare-phenomena-from","title":"Language Models Learn Rare Phenomena from Less Rare Phenomena: The Case of the Missing AANNs","date":"2024-03-28","arxiv_id":"2403.19827","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-learn-rare-phenomena-from#ran","syntology_url":"https://syntology.ai/paper/2403.19827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19827"}},"official":null}},{"url":null,"slug":"soften-to-defend-towards-adversarial","title":"Soften to Defend: Towards Adversarial Robustness via Self-Guided Label Refinement","date":"2024-03-14","arxiv_id":"2403.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"ethos-rectifying-language-models-in","title":"Ethos: Rectifying Language Models in Orthogonal Parameter Space","date":"2024-03-13","arxiv_id":"2403.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-oriented-retrieval-tuner","title":"LLM-Oriented Retrieval Tuner","date":"2024-03-04","arxiv_id":"2403.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"rome-memorization-insights-from-text","title":"ROME: Memorization Insights from Text, Logits and Representation","date":"2024-03-01","arxiv_id":"2403.00510","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-associative-memories-with-gradient","title":"Learning Associative Memories with Gradient Descent","date":"2024-02-28","arxiv_id":"2402.18724","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-privacy-memorization-and-input","title":"Unveiling Privacy, Memorization, and Input Curvature Links","date":"2024-02-28","arxiv_id":"2402.18726","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-view-of-grokking-double-descent-and","title":"Unified View of Grokking, Double Descent and Emergent Abilities: A Perspective from Circuits Competition","date":"2024-02-23","arxiv_id":"2402.15175","repositories_listed":0,"syntology":null},{"url":null,"slug":"opening-the-black-box-of-large-language","title":"Towards Uncovering How Large Language Model Works: An Explainability Perspective","date":"2024-02-16","arxiv_id":"2402.10688","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-information-organizing-and-processing","title":"Neural Information Organizing and Processing -- Neural Machines","date":"2024-02-15","arxiv_id":"2404.03676","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-complexity-of-stochastic-convex","title":"Information Complexity of Stochastic Convex Optimization: Applications to Generalization and Memorization","date":"2024-02-14","arxiv_id":"2402.09327","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-prediction-can-be-a-strong-evidence-of","title":"Future Prediction Can be a Strong Evidence of Good History Representation in Partially Observable Environments","date":"2024-02-11","arxiv_id":"2402.07102","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-evolution-of-published-text-and-the","title":"Social Evolution of Published Text and The Emergence of Artificial Intelligence Through Large Language Models and The Problem of Toxicity and Bias","date":"2024-02-11","arxiv_id":"2402.07166","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-proximal-operators-describe-score","title":"Wasserstein proximal operators describe score-based generative models and resolve memorization","date":"2024-02-09","arxiv_id":"2402.06162","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-early-learning-regularization-when","title":"Revisiting Early-Learning Regularization When Federated Learning Meets Noisy Labels","date":"2024-02-08","arxiv_id":"2402.05353","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-forgetting-advancing-machine","title":"Selective Forgetting: Advancing Machine Unlearning Techniques and Evaluation in Language Models","date":"2024-02-08","arxiv_id":"2402.05813","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-the-neural-tangent-kernel-of","title":"Analyzing the Neural Tangent Kernel of Periodically Activated Coordinate Networks","date":"2024-02-07","arxiv_id":"2402.04783","repositories_listed":0,"syntology":null},{"url":null,"slug":"brain-inspired-distributed-memorization","title":"EMN: Brain-inspired Elastic Memory Network for Quick Domain Adaptive Feature Mapping","date":"2024-02-04","arxiv_id":"2402.14598","repositories_listed":0,"syntology":null},{"url":null,"slug":"deja-vu-memorization-in-vision-language","title":"Déjà Vu Memorization in Vision-Language Models","date":"2024-02-03","arxiv_id":"2402.02103","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-centered-privacy-research-in-the-age-of","title":"Human-Centered Privacy Research in the Age of Large Language Models","date":"2024-02-03","arxiv_id":"2402.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-power-of-relu-and-step-networks","title":"Expressive Power of ReLU and Step Networks under Floating-Point Operations","date":"2024-01-26","arxiv_id":"2401.15121","repositories_listed":0,"syntology":null},{"url":null,"slug":"critical-data-size-of-language-models-from-a","title":"Critical Data Size of Language Models from a Grokking Perspective","date":"2024-01-19","arxiv_id":"2401.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-learning-through-the-lens-of","title":"Understanding Learning through the Lens of Dynamical Invariants","date":"2024-01-19","arxiv_id":"2401.10428","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-structural-labels-for-learning","title":"Learning with Structural Labels for Learning with Noisy Labels","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/noisy-correspondence-learning-with-self","slug":"noisy-correspondence-learning-with-self","title":"Noisy Correspondence Learning with Self-Reinforcing Errors Mitigation","date":"2023-12-27","arxiv_id":"2312.16478","repositories_listed":0,"syntology":null},{"url":null,"slug":"bloomvqa-assessing-hierarchical-multi-modal","title":"BloomVQA: Assessing Hierarchical Multi-modal Comprehension","date":"2023-12-20","arxiv_id":"2312.12716","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-learning-towards-collaborative","title":"Social Learning: Towards Collaborative Learning with Large Language Models","date":"2023-12-18","arxiv_id":"2312.11441","repositories_listed":0,"syntology":null},{"url":null,"slug":"emph-lifted-rdt-based-capacity-analysis-of","title":"\\emph{Lifted} RDT based capacity analysis of the 1-hidden layer treelike \\emph{sign} perceptrons neural networks","date":"2023-12-13","arxiv_id":"2312.08257","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-triggers-unveiling-memorization-in","title":"Memory Triggers: Unveiling Memorization in Text-To-Image Generative Models through Word-Level Duplication","date":"2023-12-06","arxiv_id":"2312.03692","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-un-intended-memorization-in","title":"Understanding (Un)Intended Memorization in Text-to-Image Generative Models","date":"2023-12-06","arxiv_id":"2312.07550","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-extraction-of-training-data-from","title":"Scalable Extraction of Training Data from (Production) Language Models","date":"2023-11-28","arxiv_id":"2311.17035","repositories_listed":0,"syntology":null},{"url":null,"slug":"positional-description-matters-for","title":"Positional Description Matters for Transformers Arithmetic","date":"2023-11-22","arxiv_id":"2311.14737","repositories_listed":0,"syntology":null},{"url":null,"slug":"csgnn-conquering-noisy-node-labels-via","title":"CSGNN: Conquering Noisy Node labels via Dynamic Class-wise Selection","date":"2023-11-20","arxiv_id":"2311.11473","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-retrieval-augmentation-and-the-limitations","title":"On Retrieval Augmentation and the Limitations of Language Model Training","date":"2023-11-16","arxiv_id":"2311.09615","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-pre-trained-language-model-actually","title":"Does Pre-trained Language Model Actually Infer Unseen Links in Knowledge Graph Completion?","date":"2023-11-15","arxiv_id":"2311.09109","repositories_listed":0,"syntology":null},{"url":null,"slug":"preserving-privacy-in-gans-against-membership","title":"Preserving Privacy in GANs Against Membership Inference Attack","date":"2023-11-06","arxiv_id":"2311.03172","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-statistical-thermodynamics-of-generative","title":"The statistical thermodynamics of generative diffusion models: Phase transitions, symmetry breaking and critical instability","date":"2023-10-26","arxiv_id":"2310.17467","repositories_listed":0,"syntology":null},{"url":null,"slug":"grokking-in-linear-estimators-a-solvable","title":"Grokking in Linear Estimators -- A Solvable Model that Groks without Understanding","date":"2023-10-25","arxiv_id":"2310.16441","repositories_listed":0,"syntology":null},{"url":null,"slug":"sok-memorization-in-general-purpose-large","title":"SoK: Memorization in General-Purpose Large Language Models","date":"2023-10-24","arxiv_id":"2310.18362","repositories_listed":0,"syntology":null},{"url":null,"slug":"mope-model-perturbation-based-privacy-attacks","title":"MoPe: Model Perturbation-based Privacy Attacks on Language Models","date":"2023-10-22","arxiv_id":"2310.14369","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-hallucination-assessment-for","title":"ReEval: Automatic Hallucination Evaluation for Retrieval-Augmented Large Language Models via Transferable Adversarial Attacks","date":"2023-10-19","arxiv_id":"2310.12516","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-dynamics-of-deep-network-linear","title":"Training Dynamics of Deep Network Linear Regions","date":"2023-10-19","arxiv_id":"2310.12977","repositories_listed":0,"syntology":null},{"url":null,"slug":"unintended-memorization-in-large-asr-models","title":"Unintended Memorization in Large ASR Models, and How to Mitigate It","date":"2023-10-18","arxiv_id":"2310.11739","repositories_listed":0,"syntology":null},{"url":null,"slug":"combating-label-noise-with-a-general","title":"Combating Label Noise With A General Surrogate Model For Sample Selection","date":"2023-10-16","arxiv_id":"2310.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"generation-or-replication-auscultating-audio","title":"Generation or Replication: Auscultating Audio Latent Diffusion Models","date":"2023-10-16","arxiv_id":"2310.10604","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-train-more-effective-and-efficient","title":"Why Train More? Effective and Efficient Membership Inference via Memorization","date":"2023-10-12","arxiv_id":"2310.08015","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-memorization-in-fine-tuned-language","title":"Exploring Memorization in Fine-tuned Language Models","date":"2023-10-10","arxiv_id":"2310.06714","repositories_listed":0,"syntology":null},{"url":null,"slug":"grokking-as-compression-a-nonlinear","title":"Grokking as Compression: A Nonlinear Complexity Perspective","date":"2023-10-09","arxiv_id":"2310.05918","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-do-larger-image-classifiers-memorise","title":"What do larger image classifiers memorise?","date":"2023-10-09","arxiv_id":"2310.05337","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-language-models-from-a-human","title":"Probing Large Language Models from A Human Behavioral Perspective","date":"2023-10-08","arxiv_id":"2310.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"recovery-of-training-data-from","title":"How Much Training Data is Memorized in Overparameterized Autoencoders? An Inverse Problem Perspective on Memorization Evaluation","date":"2023-10-04","arxiv_id":"2310.02897","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-associative-memories","title":"Scaling Laws for Associative Memories","date":"2023-10-04","arxiv_id":"2310.02984","repositories_listed":0,"syntology":null},{"url":"/paper/on-memorization-and-privacy-risks-of","slug":"on-memorization-and-privacy-risks-of","title":"On Memorization and Privacy Risks of Sharpness Aware Minimization","date":"2023-09-30","arxiv_id":"2310.00488","repositories_listed":0,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-memorization-and-privacy-risks-of#ran","syntology_url":"https://syntology.ai/paper/2310.00488","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00488"}},"official":null}},{"url":null,"slug":"identifying-and-mitigating-privacy-risks","title":"Identifying and Mitigating Privacy Risks Stemming from Language Models: A Survey","date":"2023-09-27","arxiv_id":"2310.01424","repositories_listed":0,"syntology":null},{"url":"/paper/learning-from-noisy-correspondence-with-tri","slug":"learning-from-noisy-correspondence-with-tri","title":"Learning From Noisy Correspondence With Tri-Partition for Cross-Modal Matching","date":"2023-09-22","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-image-transformations-facilitate","title":"Extreme Image Transformations Facilitate Robust Latent Object Representations","date":"2023-09-19","arxiv_id":"2310.07725","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-the-memorization-and","title":"Analysis of the Memorization and Generalization Capabilities of AI Agents: Are Continual Learners Robust?","date":"2023-09-18","arxiv_id":"2309.10149","repositories_listed":0,"syntology":null},{"url":null,"slug":"collectionless-artificial-intelligence","title":"Collectionless Artificial Intelligence","date":"2023-09-13","arxiv_id":"2309.06938","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-encoders-lack-knowledge-leveraging","title":"Text Encoders Lack Knowledge: Leveraging Generative LLMs for Domain-Specific Semantic Textual Similarity","date":"2023-09-12","arxiv_id":"2309.06541","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-and-attributing-the-hallucination","title":"Quantifying and Attributing the Hallucination of Large Language Models via Association Analysis","date":"2023-09-11","arxiv_id":"2309.05217","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-less-is-more-investigating-data-pruning","title":"When Less is More: Investigating Data Pruning for Pretraining LLMs at Scale","date":"2023-09-08","arxiv_id":"2309.04564","repositories_listed":0,"syntology":null},{"url":null,"slug":"flm-101b-an-open-llm-and-how-to-train-it-with","title":"FLM-101B: An Open LLM and How to Train It with $100K Budget","date":"2023-09-07","arxiv_id":"2309.03852","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-planning-search-and-memorization","title":"On the Planning, Search, and Memorization Capabilities of Large Language Models","date":"2023-09-05","arxiv_id":"2309.01868","repositories_listed":0,"syntology":null},{"url":null,"slug":"least-squares-maximum-and-weighted","title":"Least Squares Maximum and Weighted Generalization-Memorization Machines","date":"2023-08-31","arxiv_id":"2308.16456","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-and-analyzing-entity-level","title":"Quantifying and Analyzing Entity-level Memorization in Large Language Models","date":"2023-08-30","arxiv_id":"2308.15727","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-converge-toward-human","title":"Large language models converge toward human-like concept organization","date":"2023-08-29","arxiv_id":"2308.15047","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-reinforcement-learning-based","title":"Continuous Reinforcement Learning-based Dynamic Difficulty Adjustment in a Visual Working Memory Game","date":"2023-08-24","arxiv_id":"2308.12726","repositories_listed":0,"syntology":null},{"url":null,"slug":"smoothness-similarity-regularization-for-few","title":"Smoothness Similarity Regularization for Few-Shot GAN Adaptation","date":"2023-08-18","arxiv_id":"2308.09717","repositories_listed":0,"syntology":null},{"url":null,"slug":"u-turn-diffusion","title":"U-Turn Diffusion","date":"2023-08-14","arxiv_id":"2308.07421","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-e-empowering-e-commerce-authoring-with","title":"LLaMA-E: Empowering E-commerce Authoring with Object-Interleaved Instruction Following","date":"2023-08-09","arxiv_id":"2308.04913","repositories_listed":0,"syntology":null},{"url":null,"slug":"arithmetic-with-language-models-from","title":"Arithmetic with Language Models: from Memorization to Computation","date":"2023-08-02","arxiv_id":"2308.01154","repositories_listed":0,"syntology":null},{"url":null,"slug":"excitatory-inhibitory-balance-emerges-as-a","title":"Excitatory/Inhibitory Balance Emerges as a Key Factor for RBN Performance, Overriding Attractor Dynamics","date":"2023-08-02","arxiv_id":"2308.10831","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-data-protection-with-compositional","title":"Training Data Protection with Compositional Diffusion Models","date":"2023-08-02","arxiv_id":"2308.01937","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-activation-patterns-in","title":"Understanding Activation Patterns in Artificial Neural Networks by Exploring Stochastic Processes","date":"2023-08-01","arxiv_id":"2308.00858","repositories_listed":0,"syntology":null},{"url":null,"slug":"are-transformers-with-one-layer-self","title":"Are Transformers with One Layer Self-Attention Using Low-Rank Weight Matrices Universal Approximators?","date":"2023-07-26","arxiv_id":"2307.14023","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-existence-of-secret","title":"Gradient-Based Word Substitution for Obstinate Adversarial Examples Generation in Language Models","date":"2023-07-24","arxiv_id":"2307.12507","repositories_listed":0,"syntology":null},{"url":null,"slug":"distribution-shift-matters-for-knowledge","title":"Distribution Shift Matters for Knowledge Distillation with Webly Collected Images","date":"2023-07-21","arxiv_id":"2307.11469","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-can-we-learn-from-data-leakage-and","title":"What can we learn from Data Leakage and Unlearning for Law?","date":"2023-07-19","arxiv_id":"2307.10476","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-model-size-agnostic-compute-free","title":"Towards Model-Size Agnostic, Compute-Free, Memorization-based Inference of Deep Learning","date":"2023-07-14","arxiv_id":"2307.07631","repositories_listed":0,"syntology":null},{"url":null,"slug":"memorization-through-the-lens-of-curvature-of","title":"Memorization Through the Lens of Curvature of Loss Function Around Samples","date":"2023-07-11","arxiv_id":"2307.05831","repositories_listed":0,"syntology":null}],"record_sha256":"8436c8fecfdbd7ac3e5741453f3f9da06d65a5183a565755328436286cfbac45","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}