{"url":"/method/tofu","slug":"tofu","name":"Tofu","full_name":"Tofu","full_name_withheld":false,"description_markdown":"**Tofu** is an intra-layer model parallel system that partitions very large DNN models across multiple GPU devices to reduce per-GPU memory footprint. Tofu is designed to partition a dataflow graph of fine-grained tensor operators used by platforms like MXNet and TensorFlow. To optimally partition different operators in a dataflow graph, Tofu uses a recursive search algorithm that minimizes the total communication cost.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"http://arxiv.org/abs/1807.08887v2","title":"Supporting Very Large Models using Automatic Dataflow Graph Partitioning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":19,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"GUARD: Guided Unlearning and Retention via Data Attribution for Large Language Models","date":"2025-06-12","arxiv_id":"2506.10946","n_code_links":0,"syntology":null},{"paper":null,"title":"Constrained Entropic Unlearning: A Primal-Dual Framework for Large Language Models","date":"2025-06-05","arxiv_id":"2506.05314","n_code_links":0,"syntology":null},{"paper":"/paper/unierase-unlearning-token-as-a-universal","title":"UniErase: Unlearning Token as a Universal Erasure Primitive for Language Models","date":"2025-05-21","arxiv_id":"2505.15674","n_code_links":1,"syntology":null},{"paper":null,"title":"GUARD: Generation-time LLM Unlearning via Adaptive Restriction and Detection","date":"2025-05-19","arxiv_id":"2505.13312","n_code_links":0,"syntology":null},{"paper":null,"title":"UIPE: Enhancing LLM Unlearning by Removing Knowledge Related to Forgetting Targets","date":"2025-03-06","arxiv_id":"2503.04693","n_code_links":0,"syntology":null},{"paper":null,"title":"CE-U: Cross Entropy Unlearning","date":"2025-03-03","arxiv_id":"2503.01224","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-evaluation-of-unlearning-in","title":"Towards Robust Evaluation of Unlearning in LLMs via Data Transformations","date":"2024-11-23","arxiv_id":"2411.15477","n_code_links":1,"syntology":null},{"paper":null,"title":"Unlearning as multi-task optimization: A normalized gradient difference approach with an adaptive learning rate","date":"2024-10-29","arxiv_id":"2410.22086","n_code_links":0,"syntology":null},{"paper":null,"title":"LLM Unlearning via Loss Adjustment with Only Forget Data","date":"2024-10-14","arxiv_id":"2410.11143","n_code_links":0,"syntology":null},{"paper":"/paper/simplicity-prevails-rethinking-negative","title":"Simplicity Prevails: Rethinking Negative Preference Optimization for LLM Unlearning","date":"2024-10-09","arxiv_id":"2410.07163","n_code_links":2,"syntology":{"ran":8,"of":10,"unverified":2,"pointer_only":0}},{"paper":null,"title":"Answer When Needed, Forget When Not: Language Models Pretend to Forget via In-Context Knowledge Unlearning","date":"2024-10-01","arxiv_id":"2410.00382","n_code_links":0,"syntology":null},{"paper":"/paper/towards-robust-and-cost-efficient-knowledge","title":"Towards Robust and Parameter-Efficient Knowledge Unlearning for LLMs","date":"2024-08-13","arxiv_id":"2408.06621","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":0}},{"paper":"/paper/reversing-the-forget-retain-objectives-an","title":"Reversing the Forget-Retain Objectives: An Efficient LLM Unlearning Framework from Logit Difference","date":"2024-06-12","arxiv_id":"2406.08607","n_code_links":1,"syntology":null},{"paper":"/paper/negative-preference-optimization-from","title":"Negative Preference Optimization: From Catastrophic Collapse to Effective Unlearning","date":"2024-04-08","arxiv_id":"2404.05868","n_code_links":1,"syntology":null},{"paper":null,"title":"Token Fusion: Bridging the Gap between Token Pruning and Token Merging","date":"2023-12-02","arxiv_id":"2312.01026","n_code_links":0,"syntology":null},{"paper":null,"title":"On High-dimensional and Low-rank Tensor Bandits","date":"2023-05-06","arxiv_id":"2305.03884","n_code_links":0,"syntology":null},{"paper":null,"title":"TOFU: Towards Obfuscated Federated Updates by Encoding Weight Updates into Gradients from Proxy Data","date":"2022-01-21","arxiv_id":"2201.08494","n_code_links":0,"syntology":null},{"paper":null,"title":"Topologically Consistent Multi-View Face Inference Using Volumetric Sampling","date":"2021-10-06","arxiv_id":"2110.02948","n_code_links":0,"syntology":null},{"paper":null,"title":"Supporting Very Large Models using Automatic Dataflow Graph Partitioning","date":"2018-07-24","arxiv_id":"1807.08887","n_code_links":0,"syntology":null}],"papers_shown":19,"tasks":[{"task":"/task/machine-unlearning","name":"Machine Unlearning","papers":4},{"task":"/task/text-generation","name":"Text Generation","papers":2},{"task":"/task/3d-reconstruction","name":"3D Reconstruction","papers":1},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/federated-learning","name":"Federated Learning","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1},{"task":"/task/language-modeling","name":"Language Modeling","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/memorization","name":"Memorization","papers":1},{"task":"/task/model-editing","name":"Model Editing","papers":1},{"task":"/task/recommendation-systems","name":"Recommendation Systems","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/high","name":"Vocal Bursts Intensity Prediction","papers":1},{"task":"/task/world-knowledge","name":"World Knowledge","papers":1},{"task":"/task/graph-partitioning","name":"graph partitioning","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":18,"n_tasks":18,"usage_by_year":[{"year":"2018","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2023","papers":2},{"year":"2024","papers":8},{"year":"2025","papers":6}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/tofu"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}