{"as_of":"2026-08-14T09:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b37abd355759d536bfde0bc84870398cd7e522a3f82d2b4cce98c3011fbaeee2","coverage":[{"denominator":56,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:10:21.549139Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.06179/citation-record","integrity":"/paper/2506.06179/integrity","json":"/paper/2506.06179/citation-record.json","paper":"/paper/2506.06179"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.00297","last_updated":"2023-11-09T21:46:18Z","snapshot_observed_at":"2026-08-14T00:35:57.144754Z","submitted_at":"2023-06-01T02:35:57Z","title":"Transformers learn to implement preconditioned gradient descent for in-context learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00297","snapshot_observed_at":"2026-08-07T06:10:16.045295Z","title":"Transformers learn to implement preconditioned gradient descent for in-context learning, November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.045295Z"},"links":{"cited_paper":"/paper/2306.00297","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:0f1ce4b5b9193fd0e4389e4080b9ba7267bccaaa6285cfe5b25b3b01fb4814c8","observation_id":"b186e2f3-97c6-49a0-869d-367bed3f96c2","resolution":{"observed_at":"2026-08-07T06:10:16.045295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01082","last_updated":"2024-03-13T16:48:27Z","snapshot_observed_at":"2026-08-13T14:50:40.876812Z","submitted_at":"2023-10-02T10:48:42Z","title":"Linear attention is (maybe) all you need (to understand transformer optimization)","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01082","snapshot_observed_at":"2026-08-07T06:10:16.118380Z","title":"Linear attention is (maybe) all you need (to understand transformer optimization)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.118380Z"},"links":{"cited_paper":"/paper/2310.01082","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9525f766e7b7b8f0b60bc285aa426d416cad5db92ad2e0d62b4dd4076abb7bbd","observation_id":"2829d5ae-d210-43c7-a57f-8b0f2fb32faf","resolution":{"observed_at":"2026-08-07T06:10:16.118380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:25.496499Z","title":"Block coordinate descent for neural networks provably finds global minima","venue":null,"work_id":"75349709-475c-4050-a37b-7be0e47ab73a","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.308487Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:0fa752eebe55538d3fd643982ea7b5c9b2e7c0eadb89c6bb5df6e593be815092","observation_id":"b86348a3-4561-4195-b6ea-de5fbd1c5f02","resolution":{"observed_at":"2026-08-07T06:10:25.632599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04064","last_updated":"2023-10-06T07:42:39Z","snapshot_observed_at":"2026-08-13T05:55:39.216226Z","submitted_at":"2023-10-06T07:42:39Z","title":"How to Capture Higher-order Correlations? Generalizing Matrix Softmax Attention to Kronecker Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04064","snapshot_observed_at":"2026-08-07T06:10:16.430307Z","title":"and Song, Z","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.430307Z"},"links":{"cited_paper":"/paper/2310.04064","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d54bb630e8d8aab55489e7e2dfbcde1ce4df4475f4ef31c60de5448445aa0fb9","observation_id":"ed9c1163-000a-4186-bcff-5fc84f90fc4f","resolution":{"observed_at":"2026-08-07T06:10:16.430307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.0473","last_updated":"2016-05-19T21:53:22Z","snapshot_observed_at":"2026-08-12T12:07:33.202888Z","submitted_at":"2014-09-01T16:33:02Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.0473","snapshot_observed_at":"2026-08-07T06:10:16.520290Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate , May 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.520290Z"},"links":{"cited_paper":"/paper/1409.0473","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d7c5f3f6aef97822951f4a9bd969d751926f3800f0d8c2496ac4cc0880a5eca4","observation_id":"e07cc56c-df49-44ff-8f9e-95ae7a63a9e7","resolution":{"observed_at":"2026-08-07T06:10:16.520290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.11264","last_updated":"2020-10-08T12:55:37Z","snapshot_observed_at":"2026-08-08T13:16:05.781323Z","submitted_at":"2020-09-23T17:21:33Z","title":"On the Ability and Limitations of Transformers to Recognize Formal Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.11264","snapshot_observed_at":"2026-08-07T06:10:16.604932Z","title":"On the ability and limitations of transformers to recognize formal languages, 2020 a","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.604932Z"},"links":{"cited_paper":"/paper/2009.11264","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:1b3a2b5ebd7c2cf7952cd2964ff0e5276f8ab315779a689b2951e69bd3cd7acf","observation_id":"10cf59d9-00e8-4344-8bf5-a1783691c633","resolution":{"observed_at":"2026-08-07T06:10:16.604932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09286","last_updated":"2020-10-10T13:34:20Z","snapshot_observed_at":"2026-08-13T13:00:15.822430Z","submitted_at":"2020-06-16T16:27:56Z","title":"On the Computational Power of Transformers and its Implications in Sequence Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.09286","snapshot_observed_at":"2026-08-07T06:10:16.691808Z","title":"On the Computational Power of Transformers and its Implications in Sequence Modeling , October 2020 b","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.691808Z"},"links":{"cited_paper":"/paper/2006.09286","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:8ade14df7f011edf7a36217f70cd7fd3b069c0e155c49d9cd44ec710a766e85f","observation_id":"3da02f76-adfb-40e5-a345-516b08524857","resolution":{"observed_at":"2026-08-07T06:10:16.691808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:25.217859Z","title":null,"venue":null,"work_id":"a6eadd3c-4904-4060-a8a2-4837da30683c","year":1901},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.756489Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:daf8cf563b37ac4929a57f652e87072f9972f16cf775fe64afaa9329e5874668","observation_id":"7e95b74c-6263-4093-bc2a-4b82c380afc1","resolution":{"observed_at":"2026-08-07T06:10:25.347936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:16.825748Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.825748Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9ab8a524dfd7ecf7635c4ff10cae260e4ae525d5984020a30b553e20004849d7","observation_id":"d1e61cd4-1248-4fe4-9c20-fc4aaaf28195","resolution":{"observed_at":"2026-08-07T06:10:16.825748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.01345","last_updated":"2021-06-24T17:09:59Z","snapshot_observed_at":"2026-08-07T09:11:20.723647Z","submitted_at":"2021-06-02T17:53:39Z","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.01345","snapshot_observed_at":"2026-08-07T06:10:16.886542Z","title":"Decision Transformer : Reinforcement Learning via Sequence Modeling , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.886542Z"},"links":{"cited_paper":"/paper/2106.01345","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d2d7652e69c2a0cfe94592d5bb76a71340e048d2022c2dc3d7b79a2d42ed668b","observation_id":"536d2bdd-a47a-4ba5-935b-198c6aa00d06","resolution":{"observed_at":"2026-08-07T06:10:16.886542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04084","last_updated":"2024-02-06T15:39:09Z","snapshot_observed_at":"2026-08-13T04:25:05.254565Z","submitted_at":"2024-02-06T15:39:09Z","title":"Provably learning a multi-head attention layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04084","snapshot_observed_at":"2026-08-07T06:10:16.973138Z","title":"and Li, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.973138Z"},"links":{"cited_paper":"/paper/2402.04084","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:10276f0d9c311e585df19d02ef77ecc74002756806ec3831b63e5731e324e14e","observation_id":"56b1e1c8-31ae-4eb2-a06f-9c493c7e724e","resolution":{"observed_at":"2026-08-07T06:10:16.973138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19442","last_updated":"2024-06-10T17:18:07Z","snapshot_observed_at":"2026-08-13T04:06:49.870982Z","submitted_at":"2024-02-29T18:43:52Z","title":"Training Dynamics of Multi-Head Softmax Attention for In-Context Learning: Emergence, Convergence, and Optimality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19442","snapshot_observed_at":"2026-08-07T06:10:17.068890Z","title":"Training dynamics of multi-head softmax attention for in-context learning: Emergence, convergence, and optimality","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.068890Z"},"links":{"cited_paper":"/paper/2402.19442","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7e53fdd61d964f9b6bfce0d74bd5d4a0f5620e1c1c16d63a8cc7acb33f84a40d","observation_id":"3c02b589-bf51-4c11-a0ad-75d7b76b6c0f","resolution":{"observed_at":"2026-08-07T06:10:17.068890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06528","last_updated":"2024-06-04T00:20:05Z","snapshot_observed_at":"2026-08-14T08:25:53.456681Z","submitted_at":"2023-12-11T17:05:25Z","title":"Transformers Implement Functional Gradient Descent to Learn Non-Linear Functions In Context","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06528","snapshot_observed_at":"2026-08-07T06:10:17.138258Z","title":"Transformers implement functional gradient descent to learn non-linear functions in context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.138258Z"},"links":{"cited_paper":"/paper/2312.06528","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:baf388d21faf2e93d816f2cbb6cfbd6d7cd99ab78ef8af4a11e20141099299b1","observation_id":"ecfbb607-d2bd-4d8a-bb96-9b0fc2788508","resolution":{"observed_at":"2026-08-07T06:10:17.138258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.14794","last_updated":"2022-11-19T12:45:21Z","snapshot_observed_at":"2026-08-12T04:58:34.201421Z","submitted_at":"2020-09-30T17:09:09Z","title":"Rethinking Attention with Performers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.14794","snapshot_observed_at":"2026-08-07T06:10:17.231632Z","title":"Rethinking Attention with Performers , November 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.231632Z"},"links":{"cited_paper":"/paper/2009.14794","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7fc41eb9e1fc42d6d6ac2c0f335f3c6d1a4817cfd05e29e478a40e1cafd58a72","observation_id":"6a550f40-a8a8-4ee9-b9cd-7b0e0e2fd991","resolution":{"observed_at":"2026-08-07T06:10:17.231632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12680","last_updated":"2024-10-12T04:12:31Z","snapshot_observed_at":"2026-08-13T05:45:28.804765Z","submitted_at":"2023-10-19T12:18:24Z","title":"On the Optimization and Generalization of Multi-head Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12680","snapshot_observed_at":"2026-08-07T06:10:17.345237Z","title":"On the Optimization and Generalization of Multi -head Attention , October 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.345237Z"},"links":{"cited_paper":"/paper/2310.12680","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:45e5dd71d28d50f548cf09f4e3ee2facf62345ff6581549f47c4c243a0b9adc3","observation_id":"cd45c953-249f-42de-9f57-f256a71e4bd0","resolution":{"observed_at":"2026-08-07T06:10:17.345237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-07T06:10:17.408915Z","title":"BERT : Pre -training of Deep Bidirectional Transformers for Language Understanding , May 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.408915Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:386629ca7d4076ff892e909ad33bcef3f7f8e7805eaada0b72c71cbc5bd986bf","observation_id":"d9661c3a-1b2a-4d0c-a859-bb38637c1eae","resolution":{"observed_at":"2026-08-07T06:10:17.408915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T14:19:26.598265Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T06:10:17.511563Z","title":"An Image is Worth 16x16 Words : Transformers for Image Recognition at Scale , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.511563Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:083e5990c4dd1fb16664857f5c6fcef2f9d1038af91797ce645da9dab35dc038","observation_id":"f99701b9-2adf-497f-9573-589492dc5b38","resolution":{"observed_at":"2026-08-07T06:10:17.511563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.10090","last_updated":"2022-06-24T02:32:42Z","snapshot_observed_at":"2026-08-13T17:49:54.637642Z","submitted_at":"2021-10-19T16:36:19Z","title":"Inductive Biases and Variable Creation in Self-Attention Mechanisms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.10090","snapshot_observed_at":"2026-08-07T06:10:17.582759Z","title":"L., Goel, S., Kakade, S., and Zhang, C","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.582759Z"},"links":{"cited_paper":"/paper/2110.10090","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:26e6318a5adf7c4651ee57b839405c7dc10656c6ccfe003dffe92bd546ad530a","observation_id":"4d4f7717-7ac1-4bc6-b735-738760779656","resolution":{"observed_at":"2026-08-07T06:10:17.582759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:17.672307Z","title":"A mathematical framework for transformer circuits","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.672307Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7d65e4281cd97deb6aea20e41aa1b06160920f0d1eccebc2031e492a79f16b20","observation_id":"88ce5238-6b0a-4348-94a3-7f4802a095c6","resolution":{"observed_at":"2026-08-07T06:10:17.672307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-1-4471-5310-8","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"Phenotypes and Genotypes: The Search for Influential Genes, volume 18 of Computational Biology","venue":"Computational biology","work_id":"c66a3425-a4b4-4775-ad7b-15f42d1b3aff","year":2016},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.767563Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:62db4e48fd59b55862f24120f2f8df4b7eb05a9cc2f3f2fc15f5eed5930b8f01","observation_id":"31ae237e-f46c-40aa-868c-c80d2b355632","resolution":{"observed_at":"2026-08-07T06:10:22.175983Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.984804Z","title":"M., and Fan, J","venue":null,"work_id":"4fc6dafc-37d3-48f7-a800-d7cdab8678de","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.829721Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:cc5c071bf5453d1baecd27cd99ad45d37be1039463d985f98be5eebc418bef9f","observation_id":"22bbf9eb-1c4c-4bc6-92ea-a01c9e52edb3","resolution":{"observed_at":"2026-08-07T06:10:25.117893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04089","last_updated":"2024-06-06T13:59:51Z","snapshot_observed_at":"2026-08-12T23:48:45.085296Z","submitted_at":"2024-06-06T13:59:51Z","title":"On Limitation of Transformer for Learning HMMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04089","snapshot_observed_at":"2026-08-07T06:10:17.900610Z","title":"On Limitation of Transformer for Learning HMMs , June 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.900610Z"},"links":{"cited_paper":"/paper/2406.04089","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:e6458a7712d97a31770617631e4e10a0794c1545bd28b559a9bceb9256fc74a3","observation_id":"6362e547-8525-4924-bdcc-060998abbc7a","resolution":{"observed_at":"2026-08-07T06:10:17.900610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.730519Z","title":"In-context convergence of transformers","venue":null,"work_id":"918c9f14-b28a-49d0-beb9-07fd4a8d96b4","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.012937Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:8307e8982cee89e61483a9945dd664dad6a8c4eb4863d4f53cf66d5647b076da","observation_id":"7a8e3379-4fcd-4c62-a39f-9cf93381c4c3","resolution":{"observed_at":"2026-08-07T06:10:24.856363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.472918Z","title":"How Transformers Learn Diverse Attention Correlations in Masked Vision Pretraining","venue":null,"work_id":"c1636336-5e63-4e9d-81a8-6bbe52b04370","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.140528Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:0adb8cc50581dd5bce0839c56e13536dbb06820d2fe0a8e090013087840aa734","observation_id":"0f2c682a-00e9-44c6-927e-990e382e8067","resolution":{"observed_at":"2026-08-07T06:10:24.576290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09221","last_updated":"2022-10-13T19:53:56Z","snapshot_observed_at":"2026-08-13T14:05:38.444141Z","submitted_at":"2022-10-13T19:53:56Z","title":"Vision Transformers provably learn spatial structure","version":1},"cited_work":{"arxiv_id":"2210.09221","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.09221","snapshot_observed_at":"2026-08-07T06:10:23.419873Z","title":"Vision Transformers provably learn spatial structure","venue":"cs.CV","work_id":"27d0049b-8138-4a25-89a7-d12478caa341","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.267665Z"},"links":{"cited_paper":"/paper/2210.09221","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:0cffa4dc883c69e3e7d39895c1bac46d0db38d6124e39e5986f185c122cb7a94","observation_id":"a492e44c-f911-481a-84ba-313a963d527c","resolution":{"observed_at":"2026-08-07T06:10:23.476399Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1017/9781108873710","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":null,"venue":"Cambridge University Press eBooks","work_id":"26760519-728a-4e51-a628-7cac1aa2cb3c","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.493685Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:4c8b6fcfe9b394ddd4112a933f85c5b88d56da0bc3951aae0668a853280f104f","observation_id":"3bc12942-9cb9-4d32-8cc7-2ca58c640eff","resolution":{"observed_at":"2026-08-07T06:10:21.958846Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:18.579936Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.579936Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:bc9880af9f716e678b7af7a53902774447ddd6ebbb0779ac1cb8b89ca9969356","observation_id":"daa10853-5dcd-416e-a272-2b2af1fbdbb7","resolution":{"observed_at":"2026-08-07T06:10:18.579936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01655","last_updated":"2024-03-17T23:35:24Z","snapshot_observed_at":"2026-08-13T05:58:52.628139Z","submitted_at":"2023-10-02T21:39:04Z","title":"PolySketchFormer: Fast Transformers via Sketching Polynomial Kernels","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01655","snapshot_observed_at":"2026-08-07T06:10:18.702105Z","title":"PolySketchFormer : Fast Transformers via Sketching Polynomial Kernels , March 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.702105Z"},"links":{"cited_paper":"/paper/2310.01655","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:802b4a4b23a4b8b1ea1ed795b46c1730822556502695dbbead168d4dd530c54e","observation_id":"f43d11c9-bdd9-46cc-ab9f-22c1271550ef","resolution":{"observed_at":"2026-08-07T06:10:18.702105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.235812Z","title":"and Sato, I","venue":null,"work_id":"d3d91369-534b-4bc0-bd65-7f02d67f7f58","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.824938Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:86c2acb556cd41c19954693f89ef87b1dc1e7e25c82bb186a9bc323a6ef4538f","observation_id":"1f9a66d0-e2b3-4bc2-8315-fd07ad268ec9","resolution":{"observed_at":"2026-08-07T06:10:24.310452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16236","last_updated":"2020-08-31T11:09:32Z","snapshot_observed_at":"2026-08-06T23:24:28.908251Z","submitted_at":"2020-06-29T17:55:38Z","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.16236","snapshot_observed_at":"2026-08-07T06:10:18.920713Z","title":"Transformers are RNNs : Fast Autoregressive Transformers with Linear Attention","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.920713Z"},"links":{"cited_paper":"/paper/2006.16236","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:0431597402f0cdb2ca8b9bde81c5c5bc28cb65b3bd7393a5452d7f0880c14d48","observation_id":"690e44ed-757d-411e-868e-ae61c42ee29c","resolution":{"observed_at":"2026-08-07T06:10:18.920713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08898","last_updated":"2024-03-23T06:43:47Z","snapshot_observed_at":"2026-08-13T15:21:54.428549Z","submitted_at":"2022-06-17T17:15:01Z","title":"SimA: Simple Softmax-free Attention for Vision Transformers","version":2},"cited_work":{"arxiv_id":"2206.08898","doi":null,"metadata_source":"pith","pith_arxiv_id":"2206.08898","snapshot_observed_at":"2026-08-07T06:10:23.260814Z","title":"SimA: Simple Softmax-free Attention for Vision Transformers","venue":"cs.CV","work_id":"1e3e5379-20c7-4841-8a0c-b0b87224cb8a","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.033907Z"},"links":{"cited_paper":"/paper/2206.08898","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:1288392030a04470c5399428db429ec352dfed407b3ee957ad07d610ccbc8d54","observation_id":"6a9d517c-5ac8-4e1f-a088-3fbba6efc302","resolution":{"observed_at":"2026-08-07T06:10:23.317828Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.06015","last_updated":"2023-11-12T04:36:45Z","snapshot_observed_at":"2026-08-13T12:45:52.539619Z","submitted_at":"2023-02-12T22:12:35Z","title":"A Theoretical Understanding of Shallow Vision Transformers: Learning, Generalization, and Sample Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.06015","snapshot_observed_at":"2026-08-07T06:10:19.105823Z","title":"A theoretical understanding of shallow vision transformers: Learning, generalization, and sample complexity, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.105823Z"},"links":{"cited_paper":"/paper/2302.06015","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:ed34c0948bbbc360e14ed731b53ccd7b6289b74881bf264a85102a300c82bd5b","observation_id":"4e9af55a-da66-4b5e-b0a8-190a9c3b0978","resolution":{"observed_at":"2026-08-07T06:10:19.105823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.13276","last_updated":"2023-04-26T04:33:41Z","snapshot_observed_at":"2026-08-13T11:55:00.249128Z","submitted_at":"2023-04-26T04:33:41Z","title":"The Closeness of In-Context Learning and Weight Shifting for Softmax Regression","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.13276","snapshot_observed_at":"2026-08-07T06:10:19.205176Z","title":"The closeness of in-context learning and weight shifting for softmax regression","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.205176Z"},"links":{"cited_paper":"/paper/2304.13276","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a8f97fcd9876eb6cf466d73eafdf48517184fd2d7337b98a583f60c509c23506","observation_id":"10d5a435-75ec-4bac-ab9e-e3aad03ec028","resolution":{"observed_at":"2026-08-07T06:10:19.205176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.03764","last_updated":"2021-06-08T10:02:26Z","snapshot_observed_at":"2026-08-13T19:08:28.794751Z","submitted_at":"2021-06-07T16:30:28Z","title":"On the Expressive Power of Self-Attention Matrices","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.03764","snapshot_observed_at":"2026-08-07T06:10:19.335874Z","title":"On the Expressive Power of Self - Attention Matrices , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.335874Z"},"links":{"cited_paper":"/paper/2106.03764","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:617757758ccd419b581b6e41b47a0f185f47e9494fe8386baf1cbd5744541160","observation_id":"a19f856e-914a-4962-bedf-4c7694a05efa","resolution":{"observed_at":"2026-08-07T06:10:19.335874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.10749","last_updated":"2023-05-02T14:16:15Z","snapshot_observed_at":"2026-08-13T14:01:48.705870Z","submitted_at":"2022-10-19T17:45:48Z","title":"Transformers Learn Shortcuts to Automata","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.10749","snapshot_observed_at":"2026-08-07T06:10:19.422720Z","title":"T., Goel, S., Krishnamurthy, A., and Zhang, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.422720Z"},"links":{"cited_paper":"/paper/2210.10749","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:c18f38d617416d2f981ae4d531bbd4e39291e29c048bdfc220520dff87ec62c8","observation_id":"4de1ab7e-9de2-452b-9150-05f154133b75","resolution":{"observed_at":"2026-08-07T06:10:19.422720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17358","last_updated":"2024-05-30T07:54:40Z","snapshot_observed_at":"2026-08-13T14:29:52.324628Z","submitted_at":"2024-05-27T17:02:35Z","title":"Rethinking Transformers in Solving POMDPs","version":3},"cited_work":{"arxiv_id":"2405.17358","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.17358","snapshot_observed_at":"2026-08-07T06:10:22.962657Z","title":"Rethinking Transformers in Solving POMDPs","venue":"cs.LG","work_id":"e609ff2c-52c7-4244-9522-7ff7e23508b8","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.505098Z"},"links":{"cited_paper":"/paper/2405.17358","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:92cf1339480feb2ea3f3d70ea27dd45bd2084236a9871cfc1872cd3d977a25b8","observation_id":"4a00c977-f4d6-4b39-9d34-8f98f198fe77","resolution":{"observed_at":"2026-08-07T06:10:23.099785Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.009509Z","title":"Your transformer may not be as powerful as you expect","venue":null,"work_id":"df9c6213-5e6f-4eff-9813-f71d361cf811","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.623244Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9f8eb6f76e1badf75bf949862c365bb5e9c16ac53b4ad821a4ce422b014b84fb","observation_id":"609770b0-6828-4366-8570-e6ee8d580761","resolution":{"observed_at":"2026-08-07T06:10:24.114111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15478","last_updated":"2024-08-30T05:02:12Z","snapshot_observed_at":"2026-08-13T04:11:56.255894Z","submitted_at":"2024-02-23T18:12:53Z","title":"Transformers are Expressive, But Are They Expressive Enough for Regression?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15478","snapshot_observed_at":"2026-08-07T06:10:19.708151Z","title":"Transformers are expressive, but are they expressive enough for regression? arXiv preprint arXiv:2402.15478, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.708151Z"},"links":{"cited_paper":"/paper/2402.15478","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7e852c6323590187fd45a663a4c0df4ef5f51992337a16b909c0d006b34741bb","observation_id":"eccb2b48-c5b2-4ff5-abaa-3473c20fd9ba","resolution":{"observed_at":"2026-08-07T06:10:19.708151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-12T22:49:48.434250Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T06:10:19.778678Z","title":"Theory, Analysis , and Best Practices for Sigmoid Self - Attention , September 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.778678Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:1e08053e34c2d25abc8faf1990e0177e4c35bb0c2bf0a8434bf9271bdaf8ee1e","observation_id":"2737a320-d6ae-4494-b67f-3d9d3bf4cc1d","resolution":{"observed_at":"2026-08-07T06:10:19.778678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02896","last_updated":"2023-11-16T14:48:16Z","snapshot_observed_at":"2026-08-13T11:24:17.196828Z","submitted_at":"2023-06-05T14:05:04Z","title":"Representational Strengths and Limitations of Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02896","snapshot_observed_at":"2026-08-07T06:10:19.882192Z","title":"Representational Strengths and Limitations of Transformers , November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.882192Z"},"links":{"cited_paper":"/paper/2306.02896","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9e41077f3385d58d8ef030f26839c6d04a884d17710671c043dbdc6b109181d0","observation_id":"9e29a8a8-5a31-4110-b43b-ed1dbccdd87e","resolution":{"observed_at":"2026-08-07T06:10:19.882192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.05874","last_updated":"2020-09-07T21:21:37Z","snapshot_observed_at":"2026-08-14T00:38:01.295347Z","submitted_at":"2019-10-14T00:50:55Z","title":"Effects of Depth, Width, and Initialization: A Convergence Analysis of Layer-wise Training for Deep Linear Neural Networks","version":2},"cited_work":{"arxiv_id":"1910.05874","doi":"10.48550/arxiv.1910.05874","metadata_source":"pith","pith_arxiv_id":"1910.05874","snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Effects of Depth, Width, and Initialization: A Convergence Analysis of Layer-wise Training for Deep Linear Neural Networks","venue":"cs.LG","work_id":"4de9d023-cc9b-45d3-b766-92c411983919","year":2019},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.014640Z"},"links":{"cited_paper":"/paper/1910.05874","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:657383348e677b4e69e186879d10c74a39a9b7539c3d91ed71ca40e7ebd51d6d","observation_id":"a1ee5037-49bf-4eaa-844f-c146f8fa6315","resolution":{"observed_at":"2026-08-07T06:10:21.767656Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:23.802117Z","title":"Unraveling the gradient descent dynamics of transformers","venue":null,"work_id":"04d56c41-89fb-44ed-b240-60ad35088fad","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.096084Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:52ac373fd99610bd5f2b583dc61323cc187db322ac0bdffd29068cff60c8bcbf","observation_id":"db8e1eb0-9a49-4aa4-8bee-0609d5641170","resolution":{"observed_at":"2026-08-07T06:10:23.887917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.09864","last_updated":"2023-11-08T13:36:32Z","snapshot_observed_at":"2026-08-14T03:33:46.294739Z","submitted_at":"2021-04-20T09:54:06Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.09864","snapshot_observed_at":"2026-08-07T06:10:20.168393Z","title":"RoFormer : Enhanced Transformer with Rotary Position Embedding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.168393Z"},"links":{"cited_paper":"/paper/2104.09864","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d9b6838d4f1296a668a6d9c3d251a70ce7c000a3cb64bbbff9965098d747053b","observation_id":"67c703c1-8a6b-4f82-be50-7ba0e106a85f","resolution":{"observed_at":"2026-08-07T06:10:20.168393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16898","last_updated":"2024-02-22T18:38:14Z","snapshot_observed_at":"2026-08-13T10:23:06.468131Z","submitted_at":"2023-08-31T17:57:50Z","title":"Transformers as Support Vector Machines","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16898","snapshot_observed_at":"2026-08-07T06:10:20.263719Z","title":"A., Li, Y., Thrampoulidis, C., and Oymak, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.263719Z"},"links":{"cited_paper":"/paper/2308.16898","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a88a0282bac1a91661a9e4351ef7e4b0a9ea9c3b889820502493f059b8d973b1","observation_id":"2b50d96b-e00c-4ba8-9ca7-4a4b3b64e704","resolution":{"observed_at":"2026-08-07T06:10:20.263719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16380","last_updated":"2023-10-30T17:32:08Z","snapshot_observed_at":"2026-08-13T11:33:16.232315Z","submitted_at":"2023-05-25T15:59:13Z","title":"Scan and Snap: Understanding Training Dynamics and Token Composition in 1-layer Transformer","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16380","snapshot_observed_at":"2026-08-07T06:10:20.436184Z","title":"Scan and snap: Understanding training dynamics and token composition in 1-layer transformer, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.436184Z"},"links":{"cited_paper":"/paper/2305.16380","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:33d6d6393839b2fc90dad4973c0feacba84ddd911286e1c118e4bc2450cc0386","observation_id":"ec4bcdc5-53d4-47c7-a369-99fd34080d22","resolution":{"observed_at":"2026-08-07T06:10:20.436184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1501.01571","last_updated":"2015-01-07T17:46:02Z","snapshot_observed_at":"2026-08-10T11:58:59.735355Z","submitted_at":"2015-01-07T17:46:02Z","title":"An Introduction to Matrix Concentration Inequalities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1501.01571","snapshot_observed_at":"2026-08-07T06:10:20.545869Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.545869Z"},"links":{"cited_paper":"/paper/1501.01571","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:17acfbaf89c460ab7e54ef4e22bea189070eae4bcdb7c32feacd0aba6a5460fe","observation_id":"c2bc851c-fdb9-4894-8e9b-b6dbb3bdb473","resolution":{"observed_at":"2026-08-07T06:10:20.545869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-07T06:10:20.677748Z","title":"N., Kaiser, L., and Polosukhin, I","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.677748Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7f65274b3807d41e10438e553556c65feb1a3d904ac14e29d1742ec3b03102c3","observation_id":"7d115a87-ac16-4df7-9565-0ad748ae8882","resolution":{"observed_at":"2026-08-07T06:10:20.677748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.07677","last_updated":"2023-05-31T08:59:47Z","snapshot_observed_at":"2026-08-13T13:20:59.322347Z","submitted_at":"2022-12-15T09:21:21Z","title":"Transformers learn in-context by gradient descent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.07677","snapshot_observed_at":"2026-08-07T06:10:20.795541Z","title":"Transformers learn in-context by gradient descent, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.795541Z"},"links":{"cited_paper":"/paper/2212.07677","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:f00e6ad67ffcc6301c3ad1e81425d49c66545433f45cb31588646981718a621c","observation_id":"0f0c45dd-5f99-4090-af3d-5d32b5d20624","resolution":{"observed_at":"2026-08-07T06:10:20.795541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04768","last_updated":"2020-06-14T08:15:54Z","snapshot_observed_at":"2026-07-06T09:27:03.809621Z","submitted_at":"2020-06-08T17:37:52Z","title":"Linformer: Self-Attention with Linear Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04768","snapshot_observed_at":"2026-08-07T06:10:20.871924Z","title":"Z., Khabsa, M., Fang, H., and Ma, H","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.871924Z"},"links":{"cited_paper":"/paper/2006.04768","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:763d5d1d5f128c6f323d589f4fee3469ee3b3adb103504f5d5a50f597e9379f1","observation_id":"ce0005a2-a080-41ca-ba0c-86ce5636c97f","resolution":{"observed_at":"2026-08-07T06:10:20.871924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06893","last_updated":"2024-06-11T02:15:53Z","snapshot_observed_at":"2026-08-12T23:45:57.072635Z","submitted_at":"2024-06-11T02:15:53Z","title":"Transformers Provably Learn Sparse Token Selection While Fully-Connected Nets Cannot","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06893","snapshot_observed_at":"2026-08-07T06:10:20.984325Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.984325Z"},"links":{"cited_paper":"/paper/2406.06893","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a38867677e27963c77b7bbcc4768152824a6f8ca0c5a3cc4ca78563cb06b8dda","observation_id":"30eb5080-6990-49a8-882f-2efd1e737368","resolution":{"observed_at":"2026-08-07T06:10:20.984325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.13163","last_updated":"2023-03-30T06:31:06Z","snapshot_observed_at":"2026-08-13T18:38:33.552834Z","submitted_at":"2021-07-28T04:28:55Z","title":"Statistically Meaningful Approximation: a Case Study on Approximating Turing Machines with Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.13163","snapshot_observed_at":"2026-08-07T06:10:21.092976Z","title":"Statistically meaningful approximation: a case study on approximating turing machines with transformers, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.092976Z"},"links":{"cited_paper":"/paper/2107.13163","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d90f9aa1a5adc07fa3eabb153d1bd8d7e98ca4feff81625873163fd615f09266","observation_id":"4559da0e-1c67-4936-926e-00f186f34955","resolution":{"observed_at":"2026-08-07T06:10:21.092976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09605","last_updated":"2024-10-12T17:50:58Z","snapshot_observed_at":"2026-08-12T22:24:42.289389Z","submitted_at":"2024-10-12T17:50:58Z","title":"Training Dynamics of Transformers to Recognize Word Co-occurrence via Gradient Flow Analysis","version":1},"cited_work":{"arxiv_id":"2410.09605","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.09605","snapshot_observed_at":"2026-08-07T06:10:22.630026Z","title":"Training Dynamics of Transformers to Recognize Word Co-occurrence via Gradient Flow Analysis","venue":"cs.LG","work_id":"90e08731-8b4e-4450-9275-57da9a47dae1","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.173804Z"},"links":{"cited_paper":"/paper/2410.09605","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:69692f2d5c6f2a5dcb2db85490751b12cc24b4d6fe5845147bbe8c8f45ab815f","observation_id":"3c471eee-41ae-4717-a298-d15b0b9f958f","resolution":{"observed_at":"2026-08-07T06:10:22.715533Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.11115","last_updated":"2023-03-13T01:47:55Z","snapshot_observed_at":"2026-08-13T19:17:54.247861Z","submitted_at":"2021-05-24T06:42:58Z","title":"Self-Attention Networks Can Process Bounded Hierarchical Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.11115","snapshot_observed_at":"2026-08-07T06:10:21.255804Z","title":"Self-attention networks can process bounded hierarchical languages, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.255804Z"},"links":{"cited_paper":"/paper/2105.11115","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:fa6ffff636a22c1ded3aa1b94b96cac4f11ab07d6f68eac4000af4029639fd77","observation_id":"65e87cbc-8b73-4884-8d06-560981f90fff","resolution":{"observed_at":"2026-08-07T06:10:21.255804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.10077","last_updated":"2020-02-25T03:12:57Z","snapshot_observed_at":"2026-08-10T18:49:22.100973Z","submitted_at":"2019-12-20T19:49:32Z","title":"Are Transformers universal approximators of sequence-to-sequence functions?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.10077","snapshot_observed_at":"2026-08-07T06:10:21.358940Z","title":"S., Reddi, S","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.358940Z"},"links":{"cited_paper":"/paper/1912.10077","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:99f888097a5ee46559a40313ee2edeef9407c35f12354a6bdd5e13159b2f2d36","observation_id":"e11c36a9-ba77-42ee-894a-2d3458285132","resolution":{"observed_at":"2026-08-07T06:10:21.358940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.00225","last_updated":"2019-05-12T12:24:53Z","snapshot_observed_at":"2026-07-06T06:26:01.372123Z","submitted_at":"2018-03-01T06:11:53Z","title":"Global Convergence of Block Coordinate Descent in Deep Learning","version":4},"cited_work":{"arxiv_id":"1803.00225","doi":null,"metadata_source":"pith","pith_arxiv_id":"1803.00225","snapshot_observed_at":"2026-08-07T06:10:22.440807Z","title":"Global Convergence of Block Coordinate Descent in Deep Learning","venue":"math.OC","work_id":"d20c8486-399e-4adb-ba2e-360ef9251bda","year":2018},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.459603Z"},"links":{"cited_paper":"/paper/1803.00225","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:55175554df1273543b938f28079df804bfe7b0da83657f24013a290fcb38658c","observation_id":"811b1712-1d5b-43dd-8bef-2b8dfe8355f3","resolution":{"observed_at":"2026-08-07T06:10:22.512246Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.14706","last_updated":"2025-01-25T04:05:56Z","snapshot_observed_at":"2026-08-12T22:28:45.688952Z","submitted_at":"2024-10-07T20:31:13Z","title":"Transformers are Efficient Compilers, Provably","version":2},"cited_work":{"arxiv_id":"2410.14706","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.14706","snapshot_observed_at":"2026-08-07T06:10:22.288094Z","title":"Transformers are Efficient Compilers, Provably","venue":"cs.PL","work_id":"f1672e85-a282-489f-b21a-251cc56b7b06","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.549139Z"},"links":{"cited_paper":"/paper/2410.14706","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:04438e62f01be6f7c9f7bbb217085852b9317229bd0280f41e665075225bde60","observation_id":"1374c329-c8d2-4784-9261-e9b11af7be25","resolution":{"observed_at":"2026-08-07T06:10:22.360182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-12T23:28:20.349441Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization"},"reference_resolution":{"displayed":56,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":40,"verified_exact":8,"verified_fuzzy":7},"total_outbound_references":56},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 56 of 56 outbound references and 0 inbound Pith citation observations for arXiv:2506.06179."}