{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QVOEIEVVIJCU6KWMC3VP7ZG7NE","short_pith_number":"pith:QVOEIEVV","schema_version":"1.0","canonical_sha256":"855c4412b542454f2acc16eaffe4df6917de10859b66bd5ba5f7eec6f890c7eb","source":{"kind":"arxiv","id":"2011.04006","version":1},"attestation_state":"computed","paper":{"title":"Long Range Arena: A Benchmark for Efficient Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.IR"],"primary_cat":"cs.LG","authors_text":"Dara Bahri, Donald Metzler, Jinfeng Rao, Liu Yang, Mostafa Dehghani, Philip Pham, Samira Abnar, Sebastian Ruder, Yikang Shen, Yi Tay","submitted_at":"2020-11-08T15:53:56Z","abstract_excerpt":"Transformers do not scale very well to long sequence lengths largely because of quadratic self-attention complexity. In the recent months, a wide spectrum of efficient, fast Transformers have been proposed to tackle this problem, more often than not claiming superior or comparable model quality to vanilla Transformer models. To this date, there is no well-established consensus on how to evaluate this class of models. Moreover, inconsistent benchmarking on a wide spectrum of tasks and datasets makes it difficult to assess relative model quality amongst many models. This paper proposes a systema"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.04006","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-11-08T15:53:56Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.IR"],"title_canon_sha256":"e388358104aafc5f369c83eedbd7c521a6efa4132930b9b8d305e7eb689ef555","abstract_canon_sha256":"4190e18e35301043515c0a37ac3c61ee8aab2a3e29a26851379c91c123d49b45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:50:01.612985Z","signature_b64":"o5jziVU5eN0vMyAVAg+vdD6OEcseQH7M0GryQ16pMWJ4v9EMDaLP3aZdczvRtEGP5akL3PhW0ggilDewLDJlDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"855c4412b542454f2acc16eaffe4df6917de10859b66bd5ba5f7eec6f890c7eb","last_reissued_at":"2026-07-05T01:50:01.612520Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:50:01.612520Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long Range Arena: A Benchmark for Efficient Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.IR"],"primary_cat":"cs.LG","authors_text":"Dara Bahri, Donald Metzler, Jinfeng Rao, Liu Yang, Mostafa Dehghani, Philip Pham, Samira Abnar, Sebastian Ruder, Yikang Shen, Yi Tay","submitted_at":"2020-11-08T15:53:56Z","abstract_excerpt":"Transformers do not scale very well to long sequence lengths largely because of quadratic self-attention complexity. In the recent months, a wide spectrum of efficient, fast Transformers have been proposed to tackle this problem, more often than not claiming superior or comparable model quality to vanilla Transformer models. To this date, there is no well-established consensus on how to evaluate this class of models. Moreover, inconsistent benchmarking on a wide spectrum of tasks and datasets makes it difficult to assess relative model quality amongst many models. This paper proposes a systema"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.04006","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.04006/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.04006","created_at":"2026-07-05T01:50:01.612573+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.04006v1","created_at":"2026-07-05T01:50:01.612573+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.04006","created_at":"2026-07-05T01:50:01.612573+00:00"},{"alias_kind":"pith_short_12","alias_value":"QVOEIEVVIJCU","created_at":"2026-07-05T01:50:01.612573+00:00"},{"alias_kind":"pith_short_16","alias_value":"QVOEIEVVIJCU6KWM","created_at":"2026-07-05T01:50:01.612573+00:00"},{"alias_kind":"pith_short_8","alias_value":"QVOEIEVV","created_at":"2026-07-05T01:50:01.612573+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19901","citing_title":"Linear Recurrent Unit with Semantic Modulation for Image Super-Resolution","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12895","citing_title":"LongSpike: Fractional Order Spiking State Space Models for Efficient Long Sequence Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06479","citing_title":"Pretraining Recurrent Networks without Recurrence","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08171","citing_title":"Communication Dynamics Neural Networks: FFT-Diagonalized Layers for Improved Hessian Conditioning at Reduced Parameter Count","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18970","citing_title":"Advancing Intelligent Sequence Modeling: Evolution, Trade-offs, and Applications of State-Space Architectures from S4 to Mamba","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2506.04565","citing_title":"From Standalone LLMs to Integrated Intelligence: A Survey of Compound Al Systems","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06374","citing_title":"SiLIF: Structured State Space Model Dynamics and Parametrization for Spiking Neural Networks","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2511.10571","citing_title":"Differentiable Filtering for Learning Hidden Markov Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16867","citing_title":"The Falcon Series of Open Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19427","citing_title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03446","citing_title":"Fast Cross-Operator Optimization of Attention Dataflow","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00114","citing_title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08171","citing_title":"Communication Dynamics Neural Networks: FFT-Diagonalized Layers for Improved Hessian Conditioning at Reduced Parameter Count","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08539","citing_title":"Continuity Laws for Sequential Models","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22117","citing_title":"PermaFrost-Attack: Stealth Pretraining Seeding(SPS) for planting Logic Landmines During LLM Training","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10078","citing_title":"Attention-Guided Dual-Stream Learning for Group Engagement Recognition: Fusing Transformer-Encoded Motion Dynamics with Scene Context via Adaptive Gating","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14930","citing_title":"IE as Cache: Information Extraction Enhanced Agentic Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20789","citing_title":"Working Memory Constraints Scaffold Learning in Transformers under Data Scarcity","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE","json":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE.json","graph_json":"https://pith.science/api/pith-number/QVOEIEVVIJCU6KWMC3VP7ZG7NE/graph.json","events_json":"https://pith.science/api/pith-number/QVOEIEVVIJCU6KWMC3VP7ZG7NE/events.json","paper":"https://pith.science/paper/QVOEIEVV"},"agent_actions":{"view_html":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE","download_json":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE.json","view_paper":"https://pith.science/paper/QVOEIEVV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.04006&json=true","fetch_graph":"https://pith.science/api/pith-number/QVOEIEVVIJCU6KWMC3VP7ZG7NE/graph.json","fetch_events":"https://pith.science/api/pith-number/QVOEIEVVIJCU6KWMC3VP7ZG7NE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE/action/storage_attestation","attest_author":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE/action/author_attestation","sign_citation":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE/action/citation_signature","submit_replication":"https://pith.science/pith/QVOEIEVVIJCU6KWMC3VP7ZG7NE/action/replication_record"}},"created_at":"2026-07-05T01:50:01.612573+00:00","updated_at":"2026-07-05T01:50:01.612573+00:00"}