{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NHSN435BYYFLAW4CZLPQR5ZZHS","short_pith_number":"pith:NHSN435B","schema_version":"1.0","canonical_sha256":"69e4de6fa1c60ab05b82cadf08f7393c9bffa89916c6243f5cce53bbaa898f48","source":{"kind":"arxiv","id":"2210.05529","version":1},"attestation_state":"computed","paper":{"title":"An Exploration of Hierarchical Attention Transformers for Efficient Long Document Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Desmond Elliott, Ilias Chalkidis, Manos Fergadiotis, Prodromos Malakasiotis, Xiang Dai","submitted_at":"2022-10-11T15:17:56Z","abstract_excerpt":"Non-hierarchical sparse attention Transformer-based models, such as Longformer and Big Bird, are popular approaches to working with long documents. There are clear benefits to these approaches compared to the original Transformer in terms of efficiency, but Hierarchical Attention Transformer (HAT) models are a vastly understudied alternative. We develop and release fully pre-trained HAT models that use segment-wise followed by cross-segment encoders and compare them with Longformer models and partially pre-trained HATs. In several long document downstream classification tasks, our best HAT mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.05529","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-11T15:17:56Z","cross_cats_sorted":[],"title_canon_sha256":"82a4c0f7309c121f441899a037943e50f1539de588d1013f7844b3ae694ef2c7","abstract_canon_sha256":"6033cba9832494f7e51460137a0a354d71eec329b6272083df5f2eee410d5645"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:05:17.713525Z","signature_b64":"irzqV18TrJKUje3gODDtFm2wI0yVVwkzCf78V/+5p04k8IXShf2vsb98XuJRYdibXW7X2AsvZY8zQKAAG6b4Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69e4de6fa1c60ab05b82cadf08f7393c9bffa89916c6243f5cce53bbaa898f48","last_reissued_at":"2026-07-05T05:05:17.712984Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:05:17.712984Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Exploration of Hierarchical Attention Transformers for Efficient Long Document Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Desmond Elliott, Ilias Chalkidis, Manos Fergadiotis, Prodromos Malakasiotis, Xiang Dai","submitted_at":"2022-10-11T15:17:56Z","abstract_excerpt":"Non-hierarchical sparse attention Transformer-based models, such as Longformer and Big Bird, are popular approaches to working with long documents. There are clear benefits to these approaches compared to the original Transformer in terms of efficiency, but Hierarchical Attention Transformer (HAT) models are a vastly understudied alternative. We develop and release fully pre-trained HAT models that use segment-wise followed by cross-segment encoders and compare them with Longformer models and partially pre-trained HATs. In several long document downstream classification tasks, our best HAT mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.05529","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.05529/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.05529","created_at":"2026-07-05T05:05:17.713039+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.05529v1","created_at":"2026-07-05T05:05:17.713039+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.05529","created_at":"2026-07-05T05:05:17.713039+00:00"},{"alias_kind":"pith_short_12","alias_value":"NHSN435BYYFL","created_at":"2026-07-05T05:05:17.713039+00:00"},{"alias_kind":"pith_short_16","alias_value":"NHSN435BYYFLAW4C","created_at":"2026-07-05T05:05:17.713039+00:00"},{"alias_kind":"pith_short_8","alias_value":"NHSN435B","created_at":"2026-07-05T05:05:17.713039+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18525","citing_title":"Overlapping Schwarz Attention: Hierarchical Attention via Domain Decomposition","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00520","citing_title":"NEST: Nested Event Stream Transformer for Sequences of Multisets","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":235,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS","json":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS.json","graph_json":"https://pith.science/api/pith-number/NHSN435BYYFLAW4CZLPQR5ZZHS/graph.json","events_json":"https://pith.science/api/pith-number/NHSN435BYYFLAW4CZLPQR5ZZHS/events.json","paper":"https://pith.science/paper/NHSN435B"},"agent_actions":{"view_html":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS","download_json":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS.json","view_paper":"https://pith.science/paper/NHSN435B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.05529&json=true","fetch_graph":"https://pith.science/api/pith-number/NHSN435BYYFLAW4CZLPQR5ZZHS/graph.json","fetch_events":"https://pith.science/api/pith-number/NHSN435BYYFLAW4CZLPQR5ZZHS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS/action/storage_attestation","attest_author":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS/action/author_attestation","sign_citation":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS/action/citation_signature","submit_replication":"https://pith.science/pith/NHSN435BYYFLAW4CZLPQR5ZZHS/action/replication_record"}},"created_at":"2026-07-05T05:05:17.713039+00:00","updated_at":"2026-07-05T05:05:17.713039+00:00"}