{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:HTOYGQ5VIEO3UKMLT35OPYROKL","short_pith_number":"pith:HTOYGQ5V","schema_version":"1.0","canonical_sha256":"3cdd8343b5411dba298b9efae7e22e52d1eb32bccfe9225670c61cf0aff5be85","source":{"kind":"arxiv","id":"2004.10102","version":2},"attestation_state":"computed","paper":{"title":"Attention is Not Only a Weight: Analyzing Transformers with Vector Norms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Goro Kobayashi, Kentaro Inui, Sho Yokoi, Tatsuki Kuribayashi","submitted_at":"2020-04-21T15:22:27Z","abstract_excerpt":"Attention is a key component of Transformers, which have recently achieved considerable success in natural language processing. Hence, attention is being extensively studied to investigate various linguistic capabilities of Transformers, focusing on analyzing the parallels between attention weights and specific linguistic phenomena. This paper shows that attention weights alone are only one of the two factors that determine the output of attention and proposes a norm-based analysis that incorporates the second factor, the norm of the transformed input vectors. The findings of our norm-based an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.10102","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-04-21T15:22:27Z","cross_cats_sorted":[],"title_canon_sha256":"cbbdb3f9a96188cc7d2a9a856498960995e5add7a7fd9f692790e8be2987c811","abstract_canon_sha256":"c9623ddf063fada2348f06728778e4d3ee4ef77b7f496c3a73f96ebc892dce0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:40:26.819440Z","signature_b64":"G6i0UBNRunc7KsnU+bWRg9Pp8cqvoHuQ1yhdUfJyrEqdES6C/2XJ3AFdT/YkxY9HWk1CRbhJqmEtGKgtI3rrAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3cdd8343b5411dba298b9efae7e22e52d1eb32bccfe9225670c61cf0aff5be85","last_reissued_at":"2026-07-05T01:40:26.818867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:40:26.818867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attention is Not Only a Weight: Analyzing Transformers with Vector Norms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Goro Kobayashi, Kentaro Inui, Sho Yokoi, Tatsuki Kuribayashi","submitted_at":"2020-04-21T15:22:27Z","abstract_excerpt":"Attention is a key component of Transformers, which have recently achieved considerable success in natural language processing. Hence, attention is being extensively studied to investigate various linguistic capabilities of Transformers, focusing on analyzing the parallels between attention weights and specific linguistic phenomena. This paper shows that attention weights alone are only one of the two factors that determine the output of attention and proposes a norm-based analysis that incorporates the second factor, the norm of the transformed input vectors. The findings of our norm-based an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.10102","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.10102/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.10102","created_at":"2026-07-05T01:40:26.818942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.10102v2","created_at":"2026-07-05T01:40:26.818942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.10102","created_at":"2026-07-05T01:40:26.818942+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTOYGQ5VIEO3","created_at":"2026-07-05T01:40:26.818942+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTOYGQ5VIEO3UKML","created_at":"2026-07-05T01:40:26.818942+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTOYGQ5V","created_at":"2026-07-05T01:40:26.818942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13181","citing_title":"Stable Attention Response for Reliable Precipitation Nowcasting","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13258","citing_title":"Hessian-Enhanced Token Attribution (HETA): Interpreting Autoregressive LLMs","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL","json":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL.json","graph_json":"https://pith.science/api/pith-number/HTOYGQ5VIEO3UKMLT35OPYROKL/graph.json","events_json":"https://pith.science/api/pith-number/HTOYGQ5VIEO3UKMLT35OPYROKL/events.json","paper":"https://pith.science/paper/HTOYGQ5V"},"agent_actions":{"view_html":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL","download_json":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL.json","view_paper":"https://pith.science/paper/HTOYGQ5V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.10102&json=true","fetch_graph":"https://pith.science/api/pith-number/HTOYGQ5VIEO3UKMLT35OPYROKL/graph.json","fetch_events":"https://pith.science/api/pith-number/HTOYGQ5VIEO3UKMLT35OPYROKL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL/action/storage_attestation","attest_author":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL/action/author_attestation","sign_citation":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL/action/citation_signature","submit_replication":"https://pith.science/pith/HTOYGQ5VIEO3UKMLT35OPYROKL/action/replication_record"}},"created_at":"2026-07-05T01:40:26.818942+00:00","updated_at":"2026-07-05T01:40:26.818942+00:00"}