{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:W26U7XZPPU2LPIPYZUTBOAMPOW","short_pith_number":"pith:W26U7XZP","schema_version":"1.0","canonical_sha256":"b6bd4fdf2f7d34b7a1f8cd2617018f75b76ac84a9013a7dacefdd32781341d8d","source":{"kind":"arxiv","id":"2207.13259","version":1},"attestation_state":"computed","paper":{"title":"Spatiotemporal Self-attention Modeling with Temporal Patch Shift for Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Biao Wang, Chao Li, Lei Zhang, Wangmeng Xiang, Xian-Sheng Hua, Xihan Wei","submitted_at":"2022-07-27T02:47:07Z","abstract_excerpt":"Transformer-based methods have recently achieved great advancement on 2D image-based vision tasks. For 3D video-based tasks such as action recognition, however, directly applying spatiotemporal transformers on video data will bring heavy computation and memory burdens due to the largely increased number of patches and the quadratic complexity of self-attention computation. How to efficiently and effectively model the 3D self-attention of video data has been a great challenge for transformers. In this paper, we propose a Temporal Patch Shift (TPS) method for efficient 3D self-attention modeling"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.13259","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-07-27T02:47:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ad1b71343a6eba4f3ca6c4f0e42f3df00fc6b4e497fc1598fe93b54e4506ef4e","abstract_canon_sha256":"a67a906cb32bf966ef4a21e0cdc70abbd169804840579c09f0522c8fd5562c69"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:44:04.735491Z","signature_b64":"U7QYUyKJ+yzcnsaZptRX8JoflE+eepWeAX7MsfwFb6qJeEuBIj1Iuzj28qWXvU4XqDz6jwfi3/wzKA3fV/z4Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6bd4fdf2f7d34b7a1f8cd2617018f75b76ac84a9013a7dacefdd32781341d8d","last_reissued_at":"2026-07-05T04:44:04.734879Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:44:04.734879Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spatiotemporal Self-attention Modeling with Temporal Patch Shift for Action Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Biao Wang, Chao Li, Lei Zhang, Wangmeng Xiang, Xian-Sheng Hua, Xihan Wei","submitted_at":"2022-07-27T02:47:07Z","abstract_excerpt":"Transformer-based methods have recently achieved great advancement on 2D image-based vision tasks. For 3D video-based tasks such as action recognition, however, directly applying spatiotemporal transformers on video data will bring heavy computation and memory burdens due to the largely increased number of patches and the quadratic complexity of self-attention computation. How to efficiently and effectively model the 3D self-attention of video data has been a great challenge for transformers. In this paper, we propose a Temporal Patch Shift (TPS) method for efficient 3D self-attention modeling"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.13259","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.13259/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.13259","created_at":"2026-07-05T04:44:04.734953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.13259v1","created_at":"2026-07-05T04:44:04.734953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.13259","created_at":"2026-07-05T04:44:04.734953+00:00"},{"alias_kind":"pith_short_12","alias_value":"W26U7XZPPU2L","created_at":"2026-07-05T04:44:04.734953+00:00"},{"alias_kind":"pith_short_16","alias_value":"W26U7XZPPU2LPIPY","created_at":"2026-07-05T04:44:04.734953+00:00"},{"alias_kind":"pith_short_8","alias_value":"W26U7XZP","created_at":"2026-07-05T04:44:04.734953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.06783","citing_title":"Insights from Visual Cognition: Understanding Human Action Dynamics with Overall Glance and Refined Gaze Transformer","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW","json":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW.json","graph_json":"https://pith.science/api/pith-number/W26U7XZPPU2LPIPYZUTBOAMPOW/graph.json","events_json":"https://pith.science/api/pith-number/W26U7XZPPU2LPIPYZUTBOAMPOW/events.json","paper":"https://pith.science/paper/W26U7XZP"},"agent_actions":{"view_html":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW","download_json":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW.json","view_paper":"https://pith.science/paper/W26U7XZP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.13259&json=true","fetch_graph":"https://pith.science/api/pith-number/W26U7XZPPU2LPIPYZUTBOAMPOW/graph.json","fetch_events":"https://pith.science/api/pith-number/W26U7XZPPU2LPIPYZUTBOAMPOW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW/action/storage_attestation","attest_author":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW/action/author_attestation","sign_citation":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW/action/citation_signature","submit_replication":"https://pith.science/pith/W26U7XZPPU2LPIPYZUTBOAMPOW/action/replication_record"}},"created_at":"2026-07-05T04:44:04.734953+00:00","updated_at":"2026-07-05T04:44:04.734953+00:00"}