{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CEKO7UNXVXIAN3NCS4CNNTDSRQ","short_pith_number":"pith:CEKO7UNX","schema_version":"1.0","canonical_sha256":"1114efd1b7add006eda29704d6cc728c34f9fce234c588edaa615485132f531e","source":{"kind":"arxiv","id":"2205.07324","version":1},"attestation_state":"computed","paper":{"title":"Transkimmer: Transformer Learns to Layer-wise Skim","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jingwen Leng, Minyi Guo, Yue Guan, Zhengyi Li, Zhouhan Lin","submitted_at":"2022-05-15T16:23:30Z","abstract_excerpt":"Transformer architecture has become the de-facto model for many machine learning tasks from natural language processing and computer vision. As such, improving its computational efficiency becomes paramount. One of the major computational inefficiency of Transformer-based models is that they spend the identical amount of computation throughout all layers. Prior works have proposed to augment the Transformer model with the capability of skimming tokens to improve its computational efficiency. However, they suffer from not having effectual and end-to-end optimization of the discrete skimming pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.07324","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-05-15T16:23:30Z","cross_cats_sorted":[],"title_canon_sha256":"997a3f988c004aec46414fd43af9404ee2d5c652e57ffe88e1be0fda40bef4ae","abstract_canon_sha256":"009e29ce443b319488b447b3616282588215d7d5483b705c5ed2e979ff166ec4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:27.658267Z","signature_b64":"/PoZHBqJLm8oFuzCe+PnF301r9dGjtPOrapLeD3Nf+nCKaCxEQk3rOuRMhYQ2kocQxEzMeDFby9Ekx5lEJ7YBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1114efd1b7add006eda29704d6cc728c34f9fce234c588edaa615485132f531e","last_reissued_at":"2026-07-05T04:23:27.657837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:27.657837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transkimmer: Transformer Learns to Layer-wise Skim","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jingwen Leng, Minyi Guo, Yue Guan, Zhengyi Li, Zhouhan Lin","submitted_at":"2022-05-15T16:23:30Z","abstract_excerpt":"Transformer architecture has become the de-facto model for many machine learning tasks from natural language processing and computer vision. As such, improving its computational efficiency becomes paramount. One of the major computational inefficiency of Transformer-based models is that they spend the identical amount of computation throughout all layers. Prior works have proposed to augment the Transformer model with the capability of skimming tokens to improve its computational efficiency. However, they suffer from not having effectual and end-to-end optimization of the discrete skimming pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.07324","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.07324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.07324","created_at":"2026-07-05T04:23:27.657894+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.07324v1","created_at":"2026-07-05T04:23:27.657894+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.07324","created_at":"2026-07-05T04:23:27.657894+00:00"},{"alias_kind":"pith_short_12","alias_value":"CEKO7UNXVXIA","created_at":"2026-07-05T04:23:27.657894+00:00"},{"alias_kind":"pith_short_16","alias_value":"CEKO7UNXVXIAN3NC","created_at":"2026-07-05T04:23:27.657894+00:00"},{"alias_kind":"pith_short_8","alias_value":"CEKO7UNX","created_at":"2026-07-05T04:23:27.657894+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.09111","citing_title":"Reducing Reasoning Costs: The Path of Optimization for Chain of Thought via Sparse Attention Mechanism","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ","json":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ.json","graph_json":"https://pith.science/api/pith-number/CEKO7UNXVXIAN3NCS4CNNTDSRQ/graph.json","events_json":"https://pith.science/api/pith-number/CEKO7UNXVXIAN3NCS4CNNTDSRQ/events.json","paper":"https://pith.science/paper/CEKO7UNX"},"agent_actions":{"view_html":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ","download_json":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ.json","view_paper":"https://pith.science/paper/CEKO7UNX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.07324&json=true","fetch_graph":"https://pith.science/api/pith-number/CEKO7UNXVXIAN3NCS4CNNTDSRQ/graph.json","fetch_events":"https://pith.science/api/pith-number/CEKO7UNXVXIAN3NCS4CNNTDSRQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ/action/storage_attestation","attest_author":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ/action/author_attestation","sign_citation":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ/action/citation_signature","submit_replication":"https://pith.science/pith/CEKO7UNXVXIAN3NCS4CNNTDSRQ/action/replication_record"}},"created_at":"2026-07-05T04:23:27.657894+00:00","updated_at":"2026-07-05T04:23:27.657894+00:00"}