{"url":"/method/perceiver-io","slug":"perceiver-io","name":"Perceiver IO","full_name":"Perceiver IO","full_name_withheld":false,"description_markdown":"Perceiver IO is a general neural network architecture that performs well for structured input modalities and output tasks. Perceiver IO is built to easily integrate and transform arbitrary information for arbitrary tasks.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2107.14795v3","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[],"n_papers_tagged":10,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"$SE(3)$ Equivariant Ray Embeddings for Implicit Multi-View Depth Estimation","date":"2024-11-11","arxiv_id":"2411.07326","n_code_links":0,"syntology":null},{"paper":"/paper/yourmt3-multi-instrument-music-transcription","title":"YourMT3+: Multi-instrument Music Transcription with Enhanced Transformer Architectures and Cross-dataset Stem Augmentation","date":"2024-07-05","arxiv_id":"2407.04822","n_code_links":1,"syntology":null},{"paper":"/paper/fragxsitedti-revealing-responsible-segments","title":"FragXsiteDTI: Revealing Responsible Segments in Drug-Target Interaction with Transformer-Driven Interpretation","date":"2023-11-04","arxiv_id":"2311.02326","n_code_links":1,"syntology":{"ran":6,"of":8,"unverified":2,"pointer_only":8}},{"paper":"/paper/tree-cross-attention","title":"Tree Cross Attention","date":"2023-09-29","arxiv_id":"2309.17388","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":1}},{"paper":"/paper/convolution-aggregation-and-attention-based","title":"Convolution, aggregation and attention based deep neural networks for accelerating simulations in mechanics","date":"2022-12-01","arxiv_id":"2212.01386","n_code_links":1,"syntology":null},{"paper":null,"title":"Active Acquisition for Multimodal Temporal Data: A Challenging Decision-Making Task","date":"2022-11-09","arxiv_id":"2211.05039","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-speech-translation-with-dynamic","title":"Efficient Speech Translation with Dynamic Latent Perceivers","date":"2022-10-28","arxiv_id":"2210.16264","n_code_links":1,"syntology":null},{"paper":"/paper/graph-perceiver-io-a-general-architecture-for","title":"Graph Perceiver IO: A General Architecture for Graph Structured Data","date":"2022-09-14","arxiv_id":"2209.06418","n_code_links":1,"syntology":null},{"paper":null,"title":"Input-level Inductive Biases for 3D Reconstruction","date":"2021-12-06","arxiv_id":"2112.03243","n_code_links":0,"syntology":null},{"paper":"/paper/perceiver-io-a-general-architecture-for","title":"Perceiver IO: A General Architecture for Structured Inputs & Outputs","date":"2021-07-30","arxiv_id":"2107.14795","n_code_links":9,"syntology":{"ran":7,"of":11,"unverified":4,"pointer_only":0}}],"papers_shown":10,"tasks":[{"task":"/task/depth-estimation","name":"Depth Estimation","papers":2},{"task":"/task/3d-reconstruction","name":"3D Reconstruction","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/drug-discovery","name":"Drug Discovery","papers":1},{"task":"/task/drum-transcription","name":"Drum Transcription","papers":1},{"task":"/task/drum-transcription-in-music-dtm","name":"Drum Transcription in Music (DTM)","papers":1},{"task":"/task/graph-classification","name":"Graph Classification","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/informativeness","name":"Informativeness","papers":1},{"task":"/task/link-prediction","name":"Link Prediction","papers":1},{"task":"/task/multi-view-learning","name":"MULTI-VIEW LEARNING","papers":1},{"task":"/task/mixture-of-experts","name":"Mixture-of-Experts","papers":1},{"task":"/task/multi-task-learning","name":"Multi-Task Learning","papers":1},{"task":"/task/multi-instrument-music-transcription","name":"Multi-instrument Music Transcription","papers":1},{"task":"/task/music-transcription","name":"Music Transcription","papers":1},{"task":"/task/node-classification","name":"Node Classification","papers":1},{"task":"/task/optical-flow-estimation","name":"Optical Flow Estimation","papers":1}],"tasks_shown":20,"n_tasks":27,"usage_by_year":[{"year":"2021","papers":2},{"year":"2022","papers":4},{"year":"2023","papers":2},{"year":"2024","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/perceiver-io"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}