{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:V7HVNSIEVDMIQJ6UADRBFLPSCI","short_pith_number":"pith:V7HVNSIE","schema_version":"1.0","canonical_sha256":"afcf56c904a8d88827d400e212adf2122a5ed417816504edc77327320967692e","source":{"kind":"arxiv","id":"2310.10375","version":3},"attestation_state":"computed","paper":{"title":"GTA: A Geometry-Aware Attention Mechanism for Multi-View Transformers","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Andreas Geiger, Bernhard Jaeger, Max Welling, Takeru Miyato","submitted_at":"2023-10-16T13:16:09Z","abstract_excerpt":"As transformers are equivariant to the permutation of input tokens, encoding the positional information of tokens is necessary for many tasks. However, since existing positional encoding schemes have been initially designed for NLP tasks, their suitability for vision tasks, which typically exhibit different structural properties in their data, is questionable. We argue that existing positional encoding schemes are suboptimal for 3D vision tasks, as they do not respect their underlying 3D geometric structure. Based on this hypothesis, we propose a geometry-aware attention mechanism that encodes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10375","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-16T13:16:09Z","cross_cats_sorted":["cs.AI","cs.LG","stat.ML"],"title_canon_sha256":"e9eba2484dd56b41a9bdd6a3ca1fbaa97fac38940b9f2fdc365e515e9fa84472","abstract_canon_sha256":"666cb8f11835a2257d334cc8f0f77b44db12982498eaf1441cbabf7b9cd33824"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:34.371433Z","signature_b64":"Pk57lHNKwZSpeqvnCQ9trPQUHD7ArCENXKaw6WHbyvhox4TFmsiGiBD5FgnF+CLQ0940ZlF6h9AUxLpZP4iSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afcf56c904a8d88827d400e212adf2122a5ed417816504edc77327320967692e","last_reissued_at":"2026-07-05T08:28:34.370970Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:34.370970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GTA: A Geometry-Aware Attention Mechanism for Multi-View Transformers","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CV","authors_text":"Andreas Geiger, Bernhard Jaeger, Max Welling, Takeru Miyato","submitted_at":"2023-10-16T13:16:09Z","abstract_excerpt":"As transformers are equivariant to the permutation of input tokens, encoding the positional information of tokens is necessary for many tasks. However, since existing positional encoding schemes have been initially designed for NLP tasks, their suitability for vision tasks, which typically exhibit different structural properties in their data, is questionable. We argue that existing positional encoding schemes are suboptimal for 3D vision tasks, as they do not respect their underlying 3D geometric structure. Based on this hypothesis, we propose a geometry-aware attention mechanism that encodes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10375","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10375/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10375","created_at":"2026-07-05T08:28:34.371033+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10375v3","created_at":"2026-07-05T08:28:34.371033+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10375","created_at":"2026-07-05T08:28:34.371033+00:00"},{"alias_kind":"pith_short_12","alias_value":"V7HVNSIEVDMI","created_at":"2026-07-05T08:28:34.371033+00:00"},{"alias_kind":"pith_short_16","alias_value":"V7HVNSIEVDMIQJ6U","created_at":"2026-07-05T08:28:34.371033+00:00"},{"alias_kind":"pith_short_8","alias_value":"V7HVNSIE","created_at":"2026-07-05T08:28:34.371033+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05347","citing_title":"WildSplat: Feedforward Gaussian Splatting from Unposed In-the-Wild Images","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20547","citing_title":"The Token Is a Group Element: On Lie-Algebra Attention over Matrix Lie Groups","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01590","citing_title":"Effective Multi-sensor Conditioning for Street-view Novel-view Synthesis","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31585","citing_title":"DPPE: Rethinking Camera-Based Positional Encoding for Scaling Multi-View Transformers","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31336","citing_title":"DecMem: Towards Minute-Long Consistent World Generation with Decoupled Memory","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12938","citing_title":"CRePE: Curved Ray Expectation Positional Encoding for Unified-Camera-Controlled Video Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18747","citing_title":"URoPE: Universal Relative Position Embedding across Geometric Spaces","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18215","citing_title":"Memorize When Needed: Decoupled Memory Control for Spatially Consistent Long-Horizon Video Generation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI","json":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI.json","graph_json":"https://pith.science/api/pith-number/V7HVNSIEVDMIQJ6UADRBFLPSCI/graph.json","events_json":"https://pith.science/api/pith-number/V7HVNSIEVDMIQJ6UADRBFLPSCI/events.json","paper":"https://pith.science/paper/V7HVNSIE"},"agent_actions":{"view_html":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI","download_json":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI.json","view_paper":"https://pith.science/paper/V7HVNSIE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10375&json=true","fetch_graph":"https://pith.science/api/pith-number/V7HVNSIEVDMIQJ6UADRBFLPSCI/graph.json","fetch_events":"https://pith.science/api/pith-number/V7HVNSIEVDMIQJ6UADRBFLPSCI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI/action/storage_attestation","attest_author":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI/action/author_attestation","sign_citation":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI/action/citation_signature","submit_replication":"https://pith.science/pith/V7HVNSIEVDMIQJ6UADRBFLPSCI/action/replication_record"}},"created_at":"2026-07-05T08:28:34.371033+00:00","updated_at":"2026-07-05T08:28:34.371033+00:00"}