{"url":"/method/patching","slug":"patching","name":"Patching","full_name":"Activation Patching","full_name_withheld":false,"description_markdown":"Activation patching studies the model's computation by altering its latent representations, the token embeddings in transformer-based language models, during the inference process","description_state":"present","introduced_year":null,"introduced_by":{"title":"Patchscopes: A Unifying Framework for Inspecting Hidden Representations of Language Models","paper":"/paper/patchscope-a-unifying-framework-for","first_author":"Asma Ghandeharioun","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/patchscope-a-unifying-framework-for"},"source":{"url":"https://arxiv.org/abs/2401.06102v4","title":"Patchscopes: A Unifying Framework for Inspecting Hidden Representations of Language Models","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Inference Extrapolation","url":"/methods/category/inference-extrapolation","pwc_aliases":[]}],"n_papers_tagged":102,"archive_num_papers":102,"papers_newest_first":[{"paper":null,"title":"Data Augmentation in Time Series Forecasting through Inverted Framework","date":"2025-07-15","arxiv_id":"2507.11439","n_code_links":0,"syntology":null},{"paper":null,"title":"Adversarial Activation Patching: A Framework for Detecting and Mitigating Emergent Deception in Safety-Aligned Transformers","date":"2025-07-12","arxiv_id":"2507.09406","n_code_links":0,"syntology":null},{"paper":null,"title":"Unpatchable Vulnerabilities in Windows 10/11: Security Report 2025","date":"2025-07-10","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/q-resafe-assessing-safety-risks-and","title":"Q-resafe: Assessing Safety Risks and Quantization-aware Safety Patching for Quantized Large Language Models","date":"2025-06-25","arxiv_id":"2506.20251","n_code_links":1,"syntology":{"ran":2,"of":2,"unverified":0,"pointer_only":2}},{"paper":"/paper/sec-bench-automated-benchmarking-of-llm","title":"SEC-bench: Automated Benchmarking of LLM Agents on Real-World Software Security Tasks","date":"2025-06-13","arxiv_id":"2506.11791","n_code_links":1,"syntology":null},{"paper":null,"title":"Time Series Representations for Classification Lie Hidden in Pretrained Vision Transformers","date":"2025-06-10","arxiv_id":"2506.08641","n_code_links":0,"syntology":null},{"paper":"/paper/chasing-moving-targets-with-online-self-play","title":"Chasing Moving Targets with Online Self-Play Reinforcement Learning for Safer Language Models","date":"2025-06-09","arxiv_id":"2506.07468","n_code_links":1,"syntology":{"ran":6,"of":17,"unverified":11,"pointer_only":1}},{"paper":null,"title":"A Multi-Dataset Evaluation of Models for Automated Vulnerability Repair","date":"2025-06-05","arxiv_id":"2506.04987","n_code_links":0,"syntology":null},{"paper":null,"title":"Can LLMs Reason Abstractly Over Math Word Problems Without CoT? Disentangling Abstract Formulation From Arithmetic Computation","date":"2025-05-29","arxiv_id":"2505.23701","n_code_links":0,"syntology":null},{"paper":null,"title":"Foundation Model for Wireless Technology Recognition Using IQ Timeseries","date":"2025-05-26","arxiv_id":"2505.19390","n_code_links":0,"syntology":null},{"paper":"/paper/from-what-to-how-attributing-clip-s-latent","title":"From What to How: Attributing CLIP's Latent Components Reveals Unexpected Semantic Reliance","date":"2025-05-26","arxiv_id":"2505.20229","n_code_links":1,"syntology":null},{"paper":null,"title":"Co-PatcheR: Collaborative Software Patching with Component(s)-specific Small Reasoning Models","date":"2025-05-25","arxiv_id":"2505.18955","n_code_links":0,"syntology":null},{"paper":null,"title":"Know the Ropes: A Heuristic Strategy for LLM-based Multi-Agent System Design","date":"2025-05-22","arxiv_id":"2505.16979","n_code_links":0,"syntology":null},{"paper":"/paper/internal-chain-of-thought-empirical-evidence","title":"Internal Chain-of-Thought: Empirical Evidence for Layer-wise Subtask Scheduling in LLMs","date":"2025-05-20","arxiv_id":"2505.14530","n_code_links":1,"syntology":null},{"paper":null,"title":"SPIRIT: Patching Speech Language Models against Jailbreak Attacks","date":"2025-05-18","arxiv_id":"2505.13541","n_code_links":0,"syntology":null},{"paper":null,"title":"SPAT: Sensitivity-based Multihead-attention Pruning on Time Series Forecasting Models","date":"2025-05-13","arxiv_id":"2505.08768","n_code_links":0,"syntology":null},{"paper":null,"title":"Unpacking Robustness in Inflectional Languages: Adversarial Evaluation and Mechanistic Insights","date":"2025-05-08","arxiv_id":"2505.07856","n_code_links":0,"syntology":null},{"paper":"/paper/interpreting-multilingual-and-document-length","title":"Interpreting Multilingual and Document-Length Sensitive Relevance Computations in Neural Retrieval Models through Axiomatic Causal Interventions","date":"2025-05-04","arxiv_id":"2505.02154","n_code_links":1,"syntology":null},{"paper":null,"title":"The Illusion of Role Separation: Hidden Shortcuts in LLM Role Learning (and How to Fix Them)","date":"2025-05-01","arxiv_id":"2505.00626","n_code_links":0,"syntology":null},{"paper":"/paper/hse-a-plug-and-play-module-for-unified-fault","title":"HSE: A plug-and-play module for unified fault diagnosis foundation models","date":"2025-04-26","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/mib-a-mechanistic-interpretability-benchmark","title":"MIB: A Mechanistic Interpretability Benchmark","date":"2025-04-17","arxiv_id":"2504.13151","n_code_links":1,"syntology":{"ran":3,"of":3,"unverified":0,"pointer_only":0}},{"paper":"/paper/how-do-large-language-models-understand","title":"How do Large Language Models Understand Relevance? A Mechanistic Interpretability Perspective","date":"2025-04-10","arxiv_id":"2504.07898","n_code_links":1,"syntology":null},{"paper":"/paper/localized-definitions-and-distributed","title":"Localized Definitions and Distributed Reasoning: A Proof-of-Concept Mechanistic Interpretability Study via Activation Patching","date":"2025-04-03","arxiv_id":"2504.02976","n_code_links":1,"syntology":null},{"paper":"/paper/how-generative-ir-retrieves-documents","title":"Reverse-Engineering the Retrieval Process in GenIR Models","date":"2025-03-25","arxiv_id":"2503.19715","n_code_links":1,"syntology":null},{"paper":null,"title":"Sentinel: Multi-Patch Transformer with Temporal and Channel Attention for Time Series Forecasting","date":"2025-03-22","arxiv_id":"2503.17658","n_code_links":0,"syntology":null},{"paper":null,"title":"A Semantic-based Optimization Approach for Repairing LLMs: Case Study on Code Generation","date":"2025-03-17","arxiv_id":"2503.12899","n_code_links":0,"syntology":null},{"paper":null,"title":"TinySQL: A Progressive Text-to-SQL Dataset for Mechanistic Interpretability Research","date":"2025-03-17","arxiv_id":"2503.12730","n_code_links":0,"syntology":null},{"paper":null,"title":"Rethinking Lanes and Points in Complex Scenarios for Monocular 3D Lane Detection","date":"2025-03-08","arxiv_id":"2503.06237","n_code_links":0,"syntology":null},{"paper":null,"title":"TimeFound: A Foundation Model for Time Series Forecasting","date":"2025-03-06","arxiv_id":"2503.04118","n_code_links":0,"syntology":null},{"paper":"/paper/superscopes-amplifying-internal-feature","title":"Superscopes: Amplifying Internal Feature Representations for Language Model Interpretation","date":"2025-03-03","arxiv_id":"2503.02078","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/time-series-1","name":"Time Series","papers":17},{"task":"/task/time-series-forecasting","name":"Time Series Forecasting","papers":14},{"task":"/task/language-modelling","name":"Language Modelling","papers":7},{"task":"/task/language-modeling","name":"Language Modeling","papers":6},{"task":"/task/multivariate-time-series-forecasting","name":"Multivariate Time Series Forecasting","papers":6},{"task":"/task/decoder","name":"Decoder","papers":5},{"task":"/task/retrieval","name":"Retrieval","papers":5},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":4},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":4},{"task":"/task/mamba","name":"Mamba","papers":4},{"task":"/task/anomaly-detection","name":"Anomaly Detection","papers":3},{"task":"/task/eeg-1","name":"EEG","papers":3},{"task":"/task/image-classification","name":"Image Classification","papers":3},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":3},{"task":"/task/management","name":"Management","papers":3},{"task":"/task/safety-alignment","name":"Safety Alignment","papers":3},{"task":"/task/vulnerability-detection","name":"Vulnerability Detection","papers":3},{"task":"/task/image-classification","name":"image-classification","papers":3},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":2},{"task":"/task/benchmarking","name":"Benchmarking","papers":2}],"tasks_shown":20,"n_tasks":118,"usage_by_year":[{"year":"2024","papers":62},{"year":"2025","papers":40}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/patching"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}