{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3B4F2LWPJLZ6QMSUHGJTM6AMYX","short_pith_number":"pith:3B4F2LWP","schema_version":"1.0","canonical_sha256":"d8785d2ecf4af3e83254399336780cc5d314483956924f10c2096c108f0cb614","source":{"kind":"arxiv","id":"2012.15832","version":2},"attestation_state":"computed","paper":{"title":"Shortformer: Better Language Modeling using Shorter Inputs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mike Lewis, Noah A. Smith, Ofir Press","submitted_at":"2020-12-31T18:52:59Z","abstract_excerpt":"Increasing the input length has been a driver of progress in language modeling with transformers. We identify conditions where shorter inputs are not harmful, and achieve perplexity and efficiency improvements through two new methods that decrease input length. First, we show that initially training a model on short subsequences before moving on to longer ones both reduces overall training time and, surprisingly, substantially improves perplexity. Second, we show how to improve the efficiency of recurrence methods in transformers, which let models condition on previously processed tokens when "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.15832","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-12-31T18:52:59Z","cross_cats_sorted":[],"title_canon_sha256":"a69138b51e7e03a11357b83a42ea78cbe6a23ca309c7474aebb38062605f13de","abstract_canon_sha256":"763214c92bf0faf0162689489e61eff5f42b20071be35381836a77a8cc0d9d95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:45:49.293693Z","signature_b64":"BRp6r1FKp9SgcAhVkJrgIdqV0yxuhjAmZmgHFTKRgMt7IWHxIAJ1/4F7cawzCpKDsp4jvlVRjBgoiqz8KlpeBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d8785d2ecf4af3e83254399336780cc5d314483956924f10c2096c108f0cb614","last_reissued_at":"2026-07-05T02:45:49.293282Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:45:49.293282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Shortformer: Better Language Modeling using Shorter Inputs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mike Lewis, Noah A. Smith, Ofir Press","submitted_at":"2020-12-31T18:52:59Z","abstract_excerpt":"Increasing the input length has been a driver of progress in language modeling with transformers. We identify conditions where shorter inputs are not harmful, and achieve perplexity and efficiency improvements through two new methods that decrease input length. First, we show that initially training a model on short subsequences before moving on to longer ones both reduces overall training time and, surprisingly, substantially improves perplexity. Second, we show how to improve the efficiency of recurrence methods in transformers, which let models condition on previously processed tokens when "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.15832","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.15832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.15832","created_at":"2026-07-05T02:45:49.293354+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.15832v2","created_at":"2026-07-05T02:45:49.293354+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.15832","created_at":"2026-07-05T02:45:49.293354+00:00"},{"alias_kind":"pith_short_12","alias_value":"3B4F2LWPJLZ6","created_at":"2026-07-05T02:45:49.293354+00:00"},{"alias_kind":"pith_short_16","alias_value":"3B4F2LWPJLZ6QMSU","created_at":"2026-07-05T02:45:49.293354+00:00"},{"alias_kind":"pith_short_8","alias_value":"3B4F2LWP","created_at":"2026-07-05T02:45:49.293354+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2304.05969","citing_title":"Localizing Model Behavior with Path Patching","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2209.11895","citing_title":"In-context Learning and Induction Heads","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2205.01068","citing_title":"OPT: Open Pre-trained Transformer Language Models","ref_index":144,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX","json":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX.json","graph_json":"https://pith.science/api/pith-number/3B4F2LWPJLZ6QMSUHGJTM6AMYX/graph.json","events_json":"https://pith.science/api/pith-number/3B4F2LWPJLZ6QMSUHGJTM6AMYX/events.json","paper":"https://pith.science/paper/3B4F2LWP"},"agent_actions":{"view_html":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX","download_json":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX.json","view_paper":"https://pith.science/paper/3B4F2LWP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.15832&json=true","fetch_graph":"https://pith.science/api/pith-number/3B4F2LWPJLZ6QMSUHGJTM6AMYX/graph.json","fetch_events":"https://pith.science/api/pith-number/3B4F2LWPJLZ6QMSUHGJTM6AMYX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX/action/storage_attestation","attest_author":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX/action/author_attestation","sign_citation":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX/action/citation_signature","submit_replication":"https://pith.science/pith/3B4F2LWPJLZ6QMSUHGJTM6AMYX/action/replication_record"}},"created_at":"2026-07-05T02:45:49.293354+00:00","updated_at":"2026-07-05T02:45:49.293354+00:00"}