{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6JLWYSITGASIOQUKGF4W7RPNFX","short_pith_number":"pith:6JLWYSIT","schema_version":"1.0","canonical_sha256":"f2576c4913302487428a31796fc5ed2df5e9e1b82744121e408299a442453f93","source":{"kind":"arxiv","id":"2105.01601","version":4},"attestation_state":"computed","paper":{"title":"MLP-Mixer: An all-MLP Architecture for Vision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Alexey Dosovitskiy, Andreas Steiner, Daniel Keysers, Ilya Tolstikhin, Jakob Uszkoreit, Jessica Yung, Lucas Beyer, Mario Lucic, Neil Houlsby, Thomas Unterthiner, Xiaohua Zhai","submitted_at":"2021-05-04T16:17:21Z","abstract_excerpt":"Convolutional Neural Networks (CNNs) are the go-to model for computer vision. Recently, attention-based networks, such as the Vision Transformer, have also become popular. In this paper we show that while convolutions and attention are both sufficient for good performance, neither of them are necessary. We present MLP-Mixer, an architecture based exclusively on multi-layer perceptrons (MLPs). MLP-Mixer contains two types of layers: one with MLPs applied independently to image patches (i.e. \"mixing\" the per-location features), and one with MLPs applied across patches (i.e. \"mixing\" spatial info"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.01601","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-05-04T16:17:21Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"90cc52f485afceadd16ae485dcca1bf2ad0d0bf8c5428ba5d4efb83cdc678288","abstract_canon_sha256":"0d6c5c83081d7d9dd055ffaed4cf5e08f443e8eb7b9f0130ddaa3136b8755b9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:48:23.118433Z","signature_b64":"sRUsfaypx5mH3CcNOXiiKoaK5y7ykevEHBomB66PpAIQot5b3uZPqW0t+KSF33O7QbSSql/3Mrb8GDV6RUJnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2576c4913302487428a31796fc5ed2df5e9e1b82744121e408299a442453f93","last_reissued_at":"2026-07-05T02:48:23.117948Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:48:23.117948Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLP-Mixer: An all-MLP Architecture for Vision","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Alexey Dosovitskiy, Andreas Steiner, Daniel Keysers, Ilya Tolstikhin, Jakob Uszkoreit, Jessica Yung, Lucas Beyer, Mario Lucic, Neil Houlsby, Thomas Unterthiner, Xiaohua Zhai","submitted_at":"2021-05-04T16:17:21Z","abstract_excerpt":"Convolutional Neural Networks (CNNs) are the go-to model for computer vision. Recently, attention-based networks, such as the Vision Transformer, have also become popular. In this paper we show that while convolutions and attention are both sufficient for good performance, neither of them are necessary. We present MLP-Mixer, an architecture based exclusively on multi-layer perceptrons (MLPs). MLP-Mixer contains two types of layers: one with MLPs applied independently to image patches (i.e. \"mixing\" the per-location features), and one with MLPs applied across patches (i.e. \"mixing\" spatial info"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.01601","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.01601/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.01601","created_at":"2026-07-05T02:48:23.118007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.01601v4","created_at":"2026-07-05T02:48:23.118007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.01601","created_at":"2026-07-05T02:48:23.118007+00:00"},{"alias_kind":"pith_short_12","alias_value":"6JLWYSITGASI","created_at":"2026-07-05T02:48:23.118007+00:00"},{"alias_kind":"pith_short_16","alias_value":"6JLWYSITGASIOQUK","created_at":"2026-07-05T02:48:23.118007+00:00"},{"alias_kind":"pith_short_8","alias_value":"6JLWYSIT","created_at":"2026-07-05T02:48:23.118007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08175","citing_title":"Transformer-based machine learning using low-level calorimeter signals for collimated photon identification at collider experiments","ref_index":42,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25010","citing_title":"Emergent Capabilities Arise Randomly from Learning Sparse Attention Patterns","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28529","citing_title":"The Speedup Paradox: Rethinking Inference Speed-Quality Trade-off in Embodied Tasks","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28529","citing_title":"The Speedup Paradox: Rethinking Inference Speed-Quality Trade-off in Embodied Tasks","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19329","citing_title":"The Chandra-Gaia Catalog of Counterparts: Resolving ambiguous Gaia matches to X-ray sources in the Chandra Source Catalog using Machine Learning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2305.13048","citing_title":"RWKV: Reinventing RNNs for the Transformer Era","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08168","citing_title":"Understanding Asynchronous Inference Methods for Vision-Language-Action Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2111.00396","citing_title":"Efficiently Modeling Long Sequences with Structured State Spaces","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06683","citing_title":"Toeplitz MLP Mixers are Low Complexity, Information-Rich Sequence Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20934","citing_title":"SDNGuardStack: An Explainable Ensemble Learning Framework for High-Accuracy Intrusion Detection in Software-Defined Networks","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX","json":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX.json","graph_json":"https://pith.science/api/pith-number/6JLWYSITGASIOQUKGF4W7RPNFX/graph.json","events_json":"https://pith.science/api/pith-number/6JLWYSITGASIOQUKGF4W7RPNFX/events.json","paper":"https://pith.science/paper/6JLWYSIT"},"agent_actions":{"view_html":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX","download_json":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX.json","view_paper":"https://pith.science/paper/6JLWYSIT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.01601&json=true","fetch_graph":"https://pith.science/api/pith-number/6JLWYSITGASIOQUKGF4W7RPNFX/graph.json","fetch_events":"https://pith.science/api/pith-number/6JLWYSITGASIOQUKGF4W7RPNFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX/action/storage_attestation","attest_author":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX/action/author_attestation","sign_citation":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX/action/citation_signature","submit_replication":"https://pith.science/pith/6JLWYSITGASIOQUKGF4W7RPNFX/action/replication_record"}},"created_at":"2026-07-05T02:48:23.118007+00:00","updated_at":"2026-07-05T02:48:23.118007+00:00"}