{"as_of":"2026-08-11T14:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2cac69114df5967098f3c61fe44a7a15f22789900f49cbde67f47f3a177ac2cc","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:19:55.465965Z","state":"measured"},{"denominator":75,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":75,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T07:31:49.202413Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2509.22321","last_updated":"2026-04-23T14:46:53Z","snapshot_observed_at":"2026-07-06T22:30:55.313733Z","submitted_at":"2025-09-26T13:20:15Z","title":"Distributed Associative Memory via Online Convex Optimization","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-18T13:22:08.134149Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2509.22321"},"observation_digest":"sha256:b9ae6a075c1d8b111f72a40e09623a86ba9132b2d04f9f6396a8ce772897a405","observation_id":"17b1bf38-c00a-4fb3-b880-4edc9ac636c4","resolution":{"observed_at":"2026-05-18T13:22:37.598494Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2509.24552","last_updated":"2026-05-04T14:21:45Z","snapshot_observed_at":"2026-08-02T23:01:03.041288Z","submitted_at":"2025-09-29T10:04:12Z","title":"Short window attention enables long-term memorization","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-18T12:10:42.646127Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2509.24552"},"observation_digest":"sha256:46f71f29f274840c8c993f4abc02e8bc6c12206e488921794650a5f576910877","observation_id":"5ed42b36-bcf5-4ad4-8678-10b7660514b7","resolution":{"observed_at":"2026-05-18T12:11:21.723798Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":4},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-15T07:43:11.620446Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:073d7371c8c55968c519374610e0a85b640538dcf8a7af7385af1fc63fe2ed56","observation_id":"21ab0d58-e24f-4ad5-9a31-e4092c28d621","resolution":{"observed_at":"2026-05-15T07:43:11.714581Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-04T07:31:49.202413Z","title":"Understanding transformer from the perspective of associative memory.arXiv preprint arXiv:2505.19488, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.25741","last_updated":"2026-07-01T23:25:58Z","snapshot_observed_at":"2026-08-04T07:30:51.188041Z","submitted_at":"2025-10-29T17:45:42Z","title":"Scaling Latent Reasoning via Looped Language Models","version":5},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-04T07:31:49.202413Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2510.25741"},"observation_digest":"sha256:e241fd87376905f02d93e9fe46c2c3b992737e44fd03cb45eff7241e57229be9","observation_id":"625b05b7-d75d-4563-b9e1-1400f455b880","resolution":{"observed_at":"2026-08-04T07:31:49.202413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2510.26692","last_updated":"2025-11-01T12:05:18Z","snapshot_observed_at":"2026-08-07T19:30:34.681869Z","submitted_at":"2025-10-30T16:59:43Z","title":"Kimi Linear: An Expressive, Efficient Attention Architecture","version":2},"reference_index":128,"source":"pdf_text","source_observed_at":"2026-05-13T23:49:10.555255Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2510.26692"},"observation_digest":"sha256:3a7b908a3fb2562939eed3f1d276f1c4ad9eb77fd482e86e0746aa4f3f251096","observation_id":"11f08202-9af9-43b8-897b-cbf726f15efc","resolution":{"observed_at":"2026-05-13T23:49:10.979178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-03T19:39:07.995630Z","title":"Understanding trans- former from the perspective of associative memory,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.23347","last_updated":"2026-07-07T19:07:00Z","snapshot_observed_at":"2026-08-07T09:17:16.665388Z","submitted_at":"2025-11-28T16:56:18Z","title":"Distributed Dynamic Associative Memory via Online Convex Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T19:39:07.995630Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2511.23347"},"observation_digest":"sha256:a07c0563afa420e727f9c9e5b9b556a7e20c2fc464d94a8ec7c4c8200a0e1c7c","observation_id":"7847bc42-780b-4ca8-9d96-82177ccb22e6","resolution":{"observed_at":"2026-08-03T19:39:07.995630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-03T04:55:44.318733Z","title":"Zhong, S., Xu, M., Ao, T., and Shi, G","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.03681","last_updated":"2026-06-02T17:25:15Z","snapshot_observed_at":"2026-08-11T02:19:25.991798Z","submitted_at":"2026-02-03T16:02:50Z","title":"Neural Attention Search Linear: Towards Adaptive Token-Level Hybrid Attention Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T04:55:44.318733Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2602.03681"},"observation_digest":"sha256:951ecca703e31ed8eb095cb0e264d4f40b4e83106d7f17f2562caee13645ccd3","observation_id":"718c94f9-5c6e-4b86-b1fa-dab0e68b3010","resolution":{"observed_at":"2026-08-03T04:55:44.318733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2602.21204","last_updated":"2026-05-12T22:37:21Z","snapshot_observed_at":"2026-08-11T13:53:26.662684Z","submitted_at":"2026-02-24T18:59:30Z","title":"Test-Time Training with KV Binding Is Secretly Linear Attention","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T19:40:35.854519Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2602.21204"},"observation_digest":"sha256:21ea6937d5931351ad863feaa01164413a68da175f4742af47e2ae9d0b339395","observation_id":"d8aaf4b9-fbf0-4824-b867-fe7b307d8635","resolution":{"observed_at":"2026-05-15T19:41:32.692482Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-02T20:14:05.084993Z","title":"Understanding transformer from the perspective of associativememory.arXiv preprint arXiv:2505.19488, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.00198","last_updated":"2026-07-01T00:40:56Z","snapshot_observed_at":"2026-08-08T00:25:26.077628Z","submitted_at":"2026-02-27T08:11:06Z","title":"Stateful Token Reduction for Long-Video Hybrid VLMs","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T20:14:05.084993Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2603.00198"},"observation_digest":"sha256:b4ae336f52ad6561b3b3820d5fedfe1657d8e46d86245a0b1b277252c5583667","observation_id":"7c3d4610-5dc0-43d5-97c0-ff5fdd3308eb","resolution":{"observed_at":"2026-08-02T20:14:05.084993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2603.15031","last_updated":"2026-03-16T09:32:21Z","snapshot_observed_at":"2026-08-02T08:46:00.749789Z","submitted_at":"2026-03-16T09:32:21Z","title":"Attention Residuals","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-21T06:39:04.312270Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2603.15031"},"observation_digest":"sha256:a2a29145c4d6256458cec5a5f64646ed4c7d547cfa7aa748c25815a9d587a9a7","observation_id":"2bb513d6-b13d-4e65-b9ec-b5a473f04874","resolution":{"observed_at":"2026-05-21T06:39:04.482086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2605.05838","last_updated":"2026-05-07T08:12:09Z","snapshot_observed_at":"2026-08-11T13:03:21.866594Z","submitted_at":"2026-05-07T08:12:09Z","title":"MDN: Parallelizing Stepwise Momentum for Delta Linear Attention","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-09T15:27:55.566795Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2605.05838"},"observation_digest":"sha256:60ae317a6a5004506ac995742808fcf7c7bd6202c7ac5961bbd84c50f0f30866","observation_id":"319df3c7-82a0-4bef-846f-2a325020f2b7","resolution":{"observed_at":"2026-05-11T16:41:11.992338Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2605.06665","last_updated":"2026-05-07T17:59:44Z","snapshot_observed_at":"2026-08-11T05:00:15.589067Z","submitted_at":"2026-05-07T17:59:44Z","title":"UniPool: A Globally Shared Expert Pool for Mixture-of-Experts","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-08T11:56:21.623709Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2605.06665"},"observation_digest":"sha256:1374ce24faa563b8e4dada7da469b08ae4294f0a4ad9c87df3ff38595b22d017","observation_id":"c8f3772a-a66b-4470-82c4-12428b7ac48b","resolution":{"observed_at":"2026-05-11T19:26:09.342589Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2605.26647","last_updated":"2026-05-26T07:30:53Z","snapshot_observed_at":"2026-08-06T11:53:53.007749Z","submitted_at":"2026-05-26T07:30:53Z","title":"More Expressive Feedforward Layers: Part I. Token-Adaptive Mixing of Activations","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-29T19:37:51.563121Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2605.26647"},"observation_digest":"sha256:783e60cf6b59de68818d268b736b650c531deb8c0c75806fe48e3f5797647232","observation_id":"3920d616-accf-4d3d-9b74-c6a7c7f62e7e","resolution":{"observed_at":"2026-06-29T19:43:54.829295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2606.00620","last_updated":"2026-05-30T08:51:28Z","snapshot_observed_at":"2026-07-06T23:41:19.955034Z","submitted_at":"2026-05-30T08:51:28Z","title":"FlowNar: Scalable Streaming Narration for Long-Form Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T19:08:36.655886Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2606.00620"},"observation_digest":"sha256:88fe29abf96a81e192bd1f16a11cb6523d3c58d61ea696cbf3bb07f016e6e7c5","observation_id":"11e349c6-a2ae-4dd9-95b6-6a4b5e6acbc9","resolution":{"observed_at":"2026-06-28T19:12:34.727880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2606.06034","last_updated":"2026-06-04T11:29:05Z","snapshot_observed_at":"2026-08-07T23:39:50.561484Z","submitted_at":"2026-06-04T11:29:05Z","title":"When Good Enough Is Optimal: Multiplication-Only Matrix Inversion Approximation for Quantized Gated DeltaNet","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-06-28T02:36:03.243732Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2606.06034"},"observation_digest":"sha256:d194ff12345eb3776c0d126dcd6571abba655e46205a647323cb8bed75318c77","observation_id":"a5855a6b-15ae-439a-8f29-8b22ad305d86","resolution":{"observed_at":"2026-06-28T02:41:31.878765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":"2505.19488","doi":"10.48550/arxiv.2505.19488","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Understanding transformer from the perspective of as- sociative memory","venue":"ArXiv.org","work_id":"8203bff0-8a75-42c5-8c22-aa67ec7e6cf1","year":2025},"citing_paper":{"arxiv_id":"2606.31830","last_updated":"2026-06-30T15:36:48Z","snapshot_observed_at":"2026-08-03T02:38:11.037610Z","submitted_at":"2026-06-30T15:36:48Z","title":"PriorEye: Geospatial Visual Priors for End-to-End Autonomous Driving","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-07-01T05:58:58.075150Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2606.31830"},"observation_digest":"sha256:b416140c9a919c1701376b54fd8c2f8430db231e9912289a43f2d734eeaa2eef","observation_id":"b0fad31b-2c64-471d-9508-94d14446bece","resolution":{"observed_at":"2026-07-01T09:55:41.512965Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19488","snapshot_observed_at":"2026-08-01T02:44:01.805678Z","title":"Understanding transformer from the perspective of associative memory.arXiv preprint arXiv:2505.19488,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25357","last_updated":"2026-07-28T07:04:32Z","snapshot_observed_at":"2026-08-08T15:38:21.711545Z","submitted_at":"2026-07-28T07:04:32Z","title":"Raven: High-Recall Sequence Modeling with Sparse Memory Routing","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-01T02:44:01.805678Z"},"links":{"cited_paper":"/paper/2505.19488","citing_paper":"/paper/2607.25357"},"observation_digest":"sha256:eb4eb9ea009999a9aad43498c2efbacb81b3019d62e0ab8eae560397bdd9a24e","observation_id":"3108fa4a-0c09-4c80-9a01-83e56e3d107c","resolution":{"observed_at":"2026-08-01T02:44:01.805678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.19488/citation-record","integrity":"/paper/2505.19488/integrity","json":"/paper/2505.19488/citation-record.json","paper":"/paper/2505.19488"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-07T14:19:48.202553Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints.arXiv preprint arXiv:2305.13245, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.202553Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:ab2b77732e7d4c13edf51eab5f06018ce4e0088036d0001e9526aa1303954bed","observation_id":"982a9add-cfe5-4e9e-8923-37b554e4d31b","resolution":{"observed_at":"2026-08-07T14:19:48.202553Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00663","last_updated":"2024-12-31T22:32:03Z","snapshot_observed_at":"2026-08-07T09:00:33.558814Z","submitted_at":"2024-12-31T22:32:03Z","title":"Titans: Learning to Memorize at Test Time","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00663","snapshot_observed_at":"2026-08-07T14:19:48.258475Z","title":"Titans: Learning to memorize at test time.arXiv preprint arXiv:2501.00663, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.258475Z"},"links":{"cited_paper":"/paper/2501.00663","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:14f6f21235be4860dbd59427ce50cdd5c08b1f95a2e3e57361c066243ba0a216","observation_id":"84fe1e74-9ca6-4631-8a7e-e4aeb443f44d","resolution":{"observed_at":"2026-08-07T14:19:48.258475Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13173","last_updated":"2025-04-17T17:59:33Z","snapshot_observed_at":"2026-08-07T16:01:21.030997Z","submitted_at":"2025-04-17T17:59:33Z","title":"It's All Connected: A Journey Through Test-Time Memorization, Attentional Bias, Retention, and Online Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13173","snapshot_observed_at":"2026-08-07T14:19:48.338446Z","title":"It’s all connected: A journey through test-time memorization, attentional bias, retention, and online optimization, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.338446Z"},"links":{"cited_paper":"/paper/2504.13173","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:1bdbcb5619466b271987734b5d10087377b07a652693dd77ad4ab78b2f7990e9","observation_id":"072cfd15-74ff-4855-b41d-1cb5575104aa","resolution":{"observed_at":"2026-08-07T14:19:48.338446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:48.427457Z","title":"Language models are few-shot learners","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.427457Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:31e1a2b34887a9582ee6c510b43f8bae1b9df990959a529bc2ebfc11fabeef3a","observation_id":"df3a07c7-3844-4ebf-abc9-c014edbc79c5","resolution":{"observed_at":"2026-08-07T14:19:48.427457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-07T14:19:48.597642Z","title":"Flashattention-2: Faster attention with better parallelism and work partitioning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.597642Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:67de728c1db667b53af0f342e59739013e86a2133a3d69471e7b283a1cee71a8","observation_id":"59a21085-8d98-4659-93fa-8a55db78a3e2","resolution":{"observed_at":"2026-08-07T14:19:48.597642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.816940Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","venue":null,"work_id":"d8f3bd28-4d5d-4f08-a2b4-b7e9b01f0870","year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.759289Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:64eee0a46767bbedb3ed7cffc71be605ef43f41ef19b49590c1b8402baf8ab35","observation_id":"3811f605-a679-4e90-9c69-65a9ca5ed12f","resolution":{"observed_at":"2026-08-07T14:19:59.894980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-10T01:12:16.468283Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T14:19:48.858192Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale.arXiv preprint arXiv:2010.11929, 2020","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.858192Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:d7e62b390b87a8318726a1f951ba08678c723571c1aa2affb206a6bcc3952460","observation_id":"0855dbab-07a5-4ddb-968e-ddb971d1964c","resolution":{"observed_at":"2026-08-07T14:19:48.858192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.660563Z","title":"Softmax linear units.Transformer Circuits Thread, 2022","venue":null,"work_id":"5c1c9af0-b909-4f91-8cce-c55a584b8c45","year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:48.991693Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:542c0a3fa3024183bdbb5019dbaad7aa15511410b9b2f9c9e4d94aecc7c2fd2f","observation_id":"9a67a84c-89c6-41db-a10a-44decb6d1c39","resolution":{"observed_at":"2026-08-07T14:19:59.732035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.500532Z","title":"Toy models of superposition.Transformer Circuits Thread, 2022","venue":null,"work_id":"e40824e1-e352-41e8-8672-5ae2192fcdc1","year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.105811Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:25e450eee24fc7bb8c54de60bffae5fb28f4e0936e243135900e3058ac9992c6","observation_id":"948a3e61-0266-4c89-88db-0625f096b822","resolution":{"observed_at":"2026-08-07T14:19:59.576119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.350092Z","title":"Chain and causal attention for efficient entity tracking","venue":null,"work_id":"d5ddb363-cc1c-4287-aa30-4e947bfb8a25","year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.226693Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:19422b1699045ba23c4430e543fa1d21ec10ca80c95e27c87dae9ac325df2829","observation_id":"c52f9112-bd99-4065-94fa-9f191caa355c","resolution":{"observed_at":"2026-08-07T14:19:59.419926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:49.379244Z","title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity.Journal of Machine Learning Research, 23(120):1–39, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.379244Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:03fccdf8c30f340eacb93b3c83abd5668799ae202e33ce1536851780a59aff3c","observation_id":"42eafd23-1629-41d6-b6ec-0183285ccaae","resolution":{"observed_at":"2026-08-07T14:19:49.379244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.195876Z","title":"What can transformers learn in-context? a case study of simple function classes.Advancesin Neural Information Processing Systems, 35:30583–30598, 2022","venue":null,"work_id":"c8424615-aaf6-4352-825f-49bade007608","year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.531618Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:9e7cac99031e7f244e9276fdde8e1d40257ec70ba46de0f3ce5472128bb5b5ce","observation_id":"29210e14-3e92-4467-8e01-90a01f302697","resolution":{"observed_at":"2026-08-07T14:19:59.259097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.12537","last_updated":"2025-03-18T13:13:18Z","snapshot_observed_at":"2026-07-06T19:52:37.903769Z","submitted_at":"2024-11-19T14:35:38Z","title":"Unlocking State-Tracking in Linear RNNs Through Negative Eigenvalues","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.12537","snapshot_observed_at":"2026-08-07T14:19:49.695457Z","title":"Unlocking state-tracking in linear rnns through negative eigenvalues.arXiv preprint arXiv:2411.12537, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.695457Z"},"links":{"cited_paper":"/paper/2411.12537","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:e58aea29888f82c69f57251889ecd232e7a0bbe76c7c35208bd39b62ec4a5b30","observation_id":"966e8ea6-74e0-423d-9c5d-b082a58f51c0","resolution":{"observed_at":"2026-08-07T14:19:49.695457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:59.048012Z","title":"Superposition, memorization, and double descent.Transformer Circuits Thread, 6:24, 2023","venue":null,"work_id":"133f74e0-2227-47a0-9a1b-4adcbfe97f21","year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.787874Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:15436bb6044214e304f8ab59e2e25fd7007b13b87fecef5cadd769d8c27761a5","observation_id":"dd0418e1-7678-436e-ba52-c91dbc0ebcf4","resolution":{"observed_at":"2026-08-07T14:19:59.112026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.942143Z","title":"Psychology press, 2014","venue":null,"work_id":"b2ede7df-3146-4a05-91fe-ffc7b348f54a","year":2014},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:49.902093Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:878385d4324b4682d78c33b6801f04846619f749635a78d19662b807eca5a437","observation_id":"fd72a012-d7d3-444c-9c68-dada0dd9cae8","resolution":{"observed_at":"2026-08-07T14:19:58.964170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:50.067646Z","title":"Highly accurate protein structure prediction with alphafold","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.067646Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:be957a92ef11d29423bd6d9ef587376e94532779f40edb925428f00b2a3bca1f","observation_id":"f68e4245-9eb3-4d56-99bb-23c829e8995c","resolution":{"observed_at":"2026-08-07T14:19:50.067646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:50.174473Z","title":"Transformers are rnns: Fast autoregressive transformers with linear attention","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.174473Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:b658a66cbb356507f009c1baff8b29126e6cfc692ca498363c667039cee6a805","observation_id":"28f1ba6c-8e12-4c9c-af19-b91e04116cbc","resolution":{"observed_at":"2026-08-07T14:19:50.174473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.772766Z","title":"The impact of positional encoding on length generalization in transformers.Advancesin Neural Information Processing Systems, 36:24892–24928, 2023","venue":null,"work_id":"d382f68e-7622-45d3-a578-79651a594144","year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.255419Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:7c895bd90e52360019a56870e970aed7d69f0ab3a32c2cf9f4bbd6dab767b324","observation_id":"5d1d0da7-ed27-460e-b3cd-f76704d04079","resolution":{"observed_at":"2026-08-07T14:19:58.837680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.597514Z","title":"Correlation matrix memories.IEEE transactions on computers, 100(4):353–359, 1972","venue":null,"work_id":"8e10cb6e-b404-488c-bf66-2dc40729b651","year":1972},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.375466Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:ab138224d342a1a6adb0d0be77da33f124b5d98c958536ed1000535c1f9efe50","observation_id":"c46b6a12-6c04-4023-927b-6b6910fdf061","resolution":{"observed_at":"2026-08-07T14:19:58.673739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08313","last_updated":"2025-01-14T18:50:05Z","snapshot_observed_at":"2026-07-06T20:21:04.675084Z","submitted_at":"2025-01-14T18:50:05Z","title":"MiniMax-01: Scaling Foundation Models with Lightning Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08313","snapshot_observed_at":"2026-08-07T14:19:50.482050Z","title":"Minimax-01: Scaling foundation models with lightning attention.arXiv preprint arXiv:2501.08313, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.482050Z"},"links":{"cited_paper":"/paper/2501.08313","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:076f40c43de6c6f1b4afbae8804a23252163ad32a8642226b29bc52709b69c65","observation_id":"9299f381-ec98-4f30-9b48-cc9f505b8784","resolution":{"observed_at":"2026-08-07T14:19:50.482050Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02130","last_updated":"2025-03-31T19:41:52Z","snapshot_observed_at":"2026-08-07T17:32:04.628362Z","submitted_at":"2025-03-03T23:35:23Z","title":"Forgetting Transformer: Softmax Attention with a Forget Gate","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02130","snapshot_observed_at":"2026-08-07T14:19:50.604148Z","title":"Forgetting transformer: Softmax attention with a forget gate.arXiv preprint arXiv:2503.02130, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.604148Z"},"links":{"cited_paper":"/paper/2503.02130","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:efe701919c2acbaaf051c41374b56854cc75e29312566faa6e2061914b5308e2","observation_id":"98ed79ce-0bf7-42e2-b619-322b8645d438","resolution":{"observed_at":"2026-08-07T14:19:50.604148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13189","last_updated":"2025-02-18T14:06:05Z","snapshot_observed_at":"2026-07-06T20:38:48.725605Z","submitted_at":"2025-02-18T14:06:05Z","title":"MoBA: Mixture of Block Attention for Long-Context LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13189","snapshot_observed_at":"2026-08-07T14:19:50.728759Z","title":"Moba: Mixture of block attention for long-context llms.arXiv preprint arXiv:2502.13189, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.728759Z"},"links":{"cited_paper":"/paper/2502.13189","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:1e171468599508da9f8285fd48b959f3cba25709b3f017bfac970cb1f08c8782","observation_id":"e9e27009-7d3b-4fa7-9d24-95d00e5c6889","resolution":{"observed_at":"2026-08-07T14:19:50.728759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:50.834567Z","title":"The parallelism tradeoff: Limitations of log-precision transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.834567Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:9f97f8cf557705114446092f622b66e800453538c5a58cb3f243b50f5db969e7","observation_id":"fc1893a3-6414-4262-8842-8de256830b1b","resolution":{"observed_at":"2026-08-07T14:19:50.834567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:50.893975Z","title":"The illusion of state in state-space models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.893975Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:767ff7590f80ef9babc8cf98e7acc2733838a5e3010d1d85500d2f165910a692","observation_id":"60b7eee0-3001-4e58-a042-ce82f1fdc818","resolution":{"observed_at":"2026-08-07T14:19:50.893975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-07T14:19:50.974591Z","title":"In-context learning and induction heads","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:50.974591Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:940702218e6c20803b68c4977fb4c709a747cc59574710b4cd9f9ae901929e6e","observation_id":"e6906ff9-ef30-44d5-b076-2216026d0035","resolution":{"observed_at":"2026-08-07T14:19:50.974591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14456","last_updated":"2025-03-30T13:46:44Z","snapshot_observed_at":"2026-08-07T16:54:15.957609Z","submitted_at":"2025-03-18T17:31:05Z","title":"RWKV-7 \"Goose\" with Expressive Dynamic State Evolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.14456","snapshot_observed_at":"2026-08-07T14:19:51.037047Z","title":"Rwkv-7\" goose\" with expressive dynamic state evolution.arXiv preprint arXiv:2503.14456, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.037047Z"},"links":{"cited_paper":"/paper/2503.14456","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:7e7ecd2c0cc2442f34faac8656c2497c739faf2432ba4834e4d88bb84cf262fc","observation_id":"21f1e89f-9ab0-4efb-b7bf-9c63b5fe5582","resolution":{"observed_at":"2026-08-07T14:19:51.037047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00071","last_updated":"2026-02-06T19:40:50Z","snapshot_observed_at":"2026-08-01T02:15:47.181936Z","submitted_at":"2023-08-31T18:18:07Z","title":"YaRN: Efficient Context Window Extension of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00071","snapshot_observed_at":"2026-08-07T14:19:51.136138Z","title":"Yarn: Efficient context window extension of large language models.arXiv preprint arXiv:2309.00071, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.136138Z"},"links":{"cited_paper":"/paper/2309.00071","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:b9a23be5770ae047ed6362d78fb5f2741109cdbbccdbef4497f8a0b78c02130b","observation_id":"4ba66522-3505-4300-b569-2385b6571fb0","resolution":{"observed_at":"2026-08-07T14:19:51.136138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.430729Z","title":"Mechanistic design and scaling of hybrid architectures","venue":null,"work_id":"66fdc176-3238-4cf4-8a38-2fa1f2a7e61e","year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.284145Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:b0174bb3870eb2684b9c6fefbbd2537cf384407aae5ec0e61e40df3fc17e38df","observation_id":"04233f8a-33dc-4283-86b4-efae21b7bb35","resolution":{"observed_at":"2026-08-07T14:19:58.491870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:51.454975Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.454975Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:0ddcb908d28daf084a039bd52367c6dd4e35cf486f9ebc3b3e162c23c9acd254","observation_id":"81fd572e-629b-4b9e-9e2c-83797eb13b88","resolution":{"observed_at":"2026-08-07T14:19:51.454975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:51.610189Z","title":"Robust speech recognition via large-scale weak supervision","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.610189Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:aed8f06b7f673aace9c15693a8089a10ba6c7239f99dc957bdd6f763d20c71e2","observation_id":"aec05bcd-a4bb-4007-adae-4b662f9c4126","resolution":{"observed_at":"2026-08-07T14:19:51.610189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2008.02217","last_updated":"2021-04-28T07:24:49Z","snapshot_observed_at":"2026-08-10T14:03:30.039538Z","submitted_at":"2020-07-16T17:52:37Z","title":"Hopfield Networks is All You Need","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2008.02217","snapshot_observed_at":"2026-08-07T14:19:51.706205Z","title":"Hopfield networks is all you need.arXiv preprint arXiv:2008.02217, 2020","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.706205Z"},"links":{"cited_paper":"/paper/2008.02217","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:51195364047ed7fb780144242480fc90e4aee45aad772a2aee8fc24009183e0f","observation_id":"92e2d798-1895-4aff-9d9d-e75ffd282b1f","resolution":{"observed_at":"2026-08-07T14:19:51.706205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.255817Z","title":"Understanding transformer reasoning capabilities via graph algorithms.Advancesin Neural Information Processing Systems, 37:78320–78370, 2024","venue":null,"work_id":"7e80bb96-81fd-4d18-9037-b6298e667904","year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.827400Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:9281c0a96b6f61d2a878833b339698a1c78317845e501e202b881940739cb839","observation_id":"69142d11-ad3a-4729-8765-ba4f5e4ba85e","resolution":{"observed_at":"2026-08-07T14:19:58.332024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:58.100206Z","title":"Linear transformers are secretly fast weight programmers","venue":null,"work_id":"dfe5a8ae-02f9-49d1-8476-0b622f031108","year":2021},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:51.953511Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:83afb06adf7a75edde1a18ab827b8b207b6475ee1bb9f8331bd6f893c8245809","observation_id":"92766f43-3dc6-4a8e-a2bb-97bdb323697a","resolution":{"observed_at":"2026-08-07T14:19:58.177648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.08608","last_updated":"2024-07-12T22:15:02Z","snapshot_observed_at":"2026-07-06T18:44:53.587276Z","submitted_at":"2024-07-11T15:44:48Z","title":"FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.08608","snapshot_observed_at":"2026-08-07T14:19:52.093623Z","title":"Flashattention-3: Fast and accurate attention with asynchrony and low-precision.arXiv preprint arXiv:2407.08608, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.093623Z"},"links":{"cited_paper":"/paper/2407.08608","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:d45dc103c3ed80e91b7f037a7e7fd7e3f5f07fc46e750ca7a46fc211a946cae8","observation_id":"fa3efda4-3172-4620-8faa-9be43670d510","resolution":{"observed_at":"2026-08-07T14:19:52.093623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-08-11T06:21:56.129166Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-07T14:19:52.175656Z","title":"Glu variants improve transformer, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.175656Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:2a1b9f5b553960540c047ab8c090794fd45ccd13c4c29656266999c6d357285e","observation_id":"3f23cfae-66c6-4932-9515-b28d324edded","resolution":{"observed_at":"2026-08-07T14:19:52.175656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-07T14:19:52.307267Z","title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer.arXiv preprintarXiv:1701.06538, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.307267Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:e96d386efc742189351b2abc7ef7d17b119820d7694a0ff6b728dfe7519b57d6","observation_id":"cf08b6c8-1c3f-496e-9aad-3bcdb9b656a0","resolution":{"observed_at":"2026-08-07T14:19:52.307267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.09456","last_updated":"2021-11-01T15:34:23Z","snapshot_observed_at":"2026-08-11T00:07:11.399065Z","submitted_at":"2021-10-18T16:47:45Z","title":"NormFormer: Improved Transformer Pretraining with Extra Normalization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.09456","snapshot_observed_at":"2026-08-07T14:19:52.425001Z","title":"Normformer: Improved transformer pretraining with extra normaliza- tion, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.425001Z"},"links":{"cited_paper":"/paper/2110.09456","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:80edc74b47cafc161d2395ec9744f9c1db9ce2f79e676d202e47c0277a9fc110","observation_id":"e6250802-0980-4c55-8a29-cf60446cfc06","resolution":{"observed_at":"2026-08-07T14:19:52.425001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:52.549499Z","title":"Deltaproduct: Improving state-tracking in linear rnns via householder products.arXiv preprint arXiv:2502.10297, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.549499Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:86ed59dc502fe17b1b19701a6cb12f1412ee7589ded8adbe898e2a4ea6e9e690","observation_id":"6eb8c1a1-4841-46e6-9b06-0958c6c76b3b","resolution":{"observed_at":"2026-08-07T14:19:52.549499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:52.703316Z","title":"Roformer: Enhanced transformer with rotary position embedding.Neurocomputing, 568:127063, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.703316Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:9820de8664da8feec0ee2fc62fdec37ef192e8c5b4b0e3ca12d21ffaa72096c2","observation_id":"719734b1-afff-42b4-9aea-d7e01ee2d3b2","resolution":{"observed_at":"2026-08-07T14:19:52.703316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08621","last_updated":"2023-08-09T08:53:08Z","snapshot_observed_at":"2026-08-02T13:21:32.251959Z","submitted_at":"2023-07-17T16:40:01Z","title":"Retentive Network: A Successor to Transformer for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08621","snapshot_observed_at":"2026-08-07T14:19:52.899830Z","title":"Retentive network: A successor to transformer for large language models.arXiv preprint arXiv:2307.08621, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:52.899830Z"},"links":{"cited_paper":"/paper/2307.08621","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:7facbb75cfc1dc27b5d1455f22a6014bd8143a3cef861aea550d82589e31cf85","observation_id":"c206d43c-6882-49c1-9f06-55b742c8184d","resolution":{"observed_at":"2026-08-07T14:19:52.899830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.943134Z","title":"Associative learning and the hippocampus","venue":null,"work_id":"b746ad8f-5a64-44bd-bb46-acb8a65e5989","year":2005},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.044653Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:4a9cc7d93bdf787dd867bd7b6d4803e0a20dc2be409b8d8e84c5493c91906549","observation_id":"f8080f62-66c3-4f4b-9c95-86dca39712f5","resolution":{"observed_at":"2026-08-07T14:19:58.018810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:53.170456Z","title":"Attention is all you need.Advances in Neural Information Processing Systems, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.170456Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:b4f1a5e2de03ccfe27183b37539a7118b0b712ae47447c218fc823ad29f762c5","observation_id":"5977f8d5-cd66-40ac-984f-4c7c6dfabc8f","resolution":{"observed_at":"2026-08-07T14:19:53.170456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:53.312631Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.312631Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:ba3bf841e1377ec8bcbdb5412d771f064fcb75754a4ba737edadea655fbb3c66","observation_id":"35f60b11-1684-4761-86d2-f3e467f9bf2f","resolution":{"observed_at":"2026-08-07T14:19:53.312631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12224","last_updated":"2024-05-28T01:38:59Z","snapshot_observed_at":"2026-07-06T18:02:15.186390Z","submitted_at":"2024-04-18T14:38:32Z","title":"Length Generalization of Causal Transformers without Position Encoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12224","snapshot_observed_at":"2026-08-07T14:19:53.463146Z","title":"Length generalization of causal transformers without position encoding.arXiv preprint arXiv:2404.12224, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.463146Z"},"links":{"cited_paper":"/paper/2404.12224","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:5b7c1ffff41dd6f82a722c87dfaec9cb70ee8571b3d973492c7b7bdc23740489","observation_id":"a7586f8e-18a6-4212-9f91-f085bd39dd63","resolution":{"observed_at":"2026-08-07T14:19:53.463146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12352","last_updated":"2025-05-02T02:07:05Z","snapshot_observed_at":"2026-08-10T17:12:18.368276Z","submitted_at":"2025-01-21T18:32:31Z","title":"Test-time regression: a unifying framework for designing sequence models with associative memory","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12352","snapshot_observed_at":"2026-08-07T14:19:53.627251Z","title":"Test-time regression: a unifying framework for designing sequence models with associative memory.arXiv preprint arXiv:2501.12352, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.627251Z"},"links":{"cited_paper":"/paper/2501.12352","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:3f5418373b18a33b1d37d03d79f1b113bd0b5bc9af61e4f71cc431f890ffd685","observation_id":"63350a0c-44c2-4415-90fd-f84177b3fd46","resolution":{"observed_at":"2026-08-07T14:19:53.627251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19574","last_updated":"2024-12-05T12:19:38Z","snapshot_observed_at":"2026-07-06T19:58:48.586415Z","submitted_at":"2024-11-29T09:42:38Z","title":"KV Shifting Attention Enhances Language Modeling","version":2},"cited_work":{"arxiv_id":"2411.19574","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.19574","snapshot_observed_at":"2026-08-07T14:19:55.642461Z","title":"KV Shifting Attention Enhances Language Modeling","venue":"cs.CL","work_id":"461d8ea4-eadf-4df0-9084-d9add26f7913","year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.772068Z"},"links":{"cited_paper":"/paper/2411.19574","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:eea55a31812b5c53f2bcdc04cd066f6fc54a036808a2210725cacd824b2e9fba","observation_id":"3faf0b22-898e-4b7a-a7ad-d4121003e275","resolution":{"observed_at":"2026-08-07T14:19:55.723952Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06635","last_updated":"2024-08-27T01:27:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-11T18:51:59Z","title":"Gated Linear Attention Transformers with Hardware-Efficient Training","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06635","snapshot_observed_at":"2026-08-07T14:19:53.893407Z","title":"Gated linear attention transformers with hardware-efficient training.arXiv preprint arXiv:2312.06635, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:53.893407Z"},"links":{"cited_paper":"/paper/2312.06635","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:c90b71bb0a430f2a11c7441e6e82bc108821357f0a5445d129fc5ed5c666ebaa","observation_id":"1fd6cf88-b12e-4883-9be2-cb53f0bd329b","resolution":{"observed_at":"2026-08-07T14:19:53.893407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06484","last_updated":"2025-01-15T10:41:40Z","snapshot_observed_at":"2026-08-06T05:53:13.494942Z","submitted_at":"2024-06-10T17:24:42Z","title":"Parallelizing Linear Transformers with the Delta Rule over Sequence Length","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06484","snapshot_observed_at":"2026-08-07T14:19:54.017185Z","title":"Parallelizing linear transformers with the delta rule over sequence length.arXiv preprint arXiv:2406.06484, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.017185Z"},"links":{"cited_paper":"/paper/2406.06484","citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:43c99e65e135f18cb7fa25d5aed8de3695abf95672e6200cfa454ebd586b9e1a","observation_id":"02c654cb-d479-431e-b9f7-197bf653f528","resolution":{"observed_at":"2026-08-07T14:19:54.017185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.786015Z","title":"Root mean square layer normalization.Advancesin Neural Information Processing Systems, 32, 2019","venue":null,"work_id":"4ce6aa94-af79-40bc-ac66-71462c29ec63","year":2019},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.205182Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:b2fc831e8fa75c4bf7523c3775290c718a58e659c5bdd76677da715736fdb4d1","observation_id":"1d7d8ae9-6785-499f-9f10-6b77be9136ce","resolution":{"observed_at":"2026-08-07T14:19:57.838596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.600762Z","title":"Probabilistic methods in combinatorics.Draft available at https://yufeizhao","venue":null,"work_id":"681ce2cf-20e5-4003-9b03-8d8bbd3227de","year":2022},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.328337Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:7f2238460c5f5d5707e8f5d919ed72b43a5cc925654453318c692597a37eade1","observation_id":"8a34f8f1-763d-45f6-aee5-0dfc94d66f9d","resolution":{"observed_at":"2026-08-07T14:19:57.696442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.440118Z","title":"\"\" n is the pr evi ou s v , v is a ct ua lly new v","venue":null,"work_id":"7a95214c-4102-4caf-ac09-bb8c3b366bda","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.423769Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:fb446d8aa944d047416b211940800bd83655d30670dc577b1796f0fef475dfce","observation_id":"5e3d3d69-df75-44da-ad7a-9653f1d784dc","resolution":{"observed_at":"2026-08-07T14:19:57.537887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.284778Z","title":null,"venue":null,"work_id":"a311e7ba-9fcf-4895-826d-5ec08c83f8c0","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.568386Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:617e1af468a9fb2296c11e98ee832a4560a8fe96c1f2ad40f8a714f721f16a33","observation_id":"925954fa-a989-4797-bda0-3d72d3878531","resolution":{"observed_at":"2026-08-07T14:19:57.356697Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:57.096462Z","title":null,"venue":null,"work_id":"2dec8788-d8a7-4487-babb-6267b5d1d34a","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.716664Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:fc821e9b6d877f120124d43049cf56af9ff5f1d15e30e12b4be146a30d0a15b1","observation_id":"cdb240bc-9b05-4561-92a3-b0daeb3c19a2","resolution":{"observed_at":"2026-08-07T14:19:57.182989Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:56.938026Z","title":null,"venue":null,"work_id":"ed9ae058-a8ba-4df5-94c9-aeec6382e35f","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:54.932721Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:633ea3c453a50ff8a0d0e244725c524da3da815af88d291a9ab227d5fe5ad5ba","observation_id":"d65f8227-e3ce-4ee7-a321-19d3b2145287","resolution":{"observed_at":"2026-08-07T14:19:57.022894Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:56.786113Z","title":null,"venue":null,"work_id":"6c7568f2-3850-43d6-ba4a-596ea4dfc90f","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:55.064815Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:36784b74cbd51d2cc743d6f47a6c11dc474692cbce2e4eb553368a9840f0eed7","observation_id":"dbaaf141-3765-4be0-9339-1588d98e27f0","resolution":{"observed_at":"2026-08-07T14:19:56.843672Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:56.589302Z","title":null,"venue":null,"work_id":"3c9bcbed-b994-4a94-a63b-44cfdb92aff8","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:55.192014Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:64df8810dce07e29319ba8dae0a1a4591b70c7c350fbc65cec54c5aa331c78fd","observation_id":"81cd2aea-298a-4191-9d65-c95d148cf0fd","resolution":{"observed_at":"2026-08-07T14:19:56.684780Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:56.387822Z","title":"Combining all the above cases, we have completed the proof","venue":null,"work_id":"ca1aca1a-e4cb-420b-b015-25b79d58de5b","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:55.311424Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:eb39046b21af08e2f2deae639e187269160cd98b038b8d7b7eac7ff479bf45cb","observation_id":"cb31a264-361d-4006-b561-8edc3aac31c6","resolution":{"observed_at":"2026-08-07T14:19:56.452139Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:19:56.238977Z","title":"b h s d , b h t d - > b h s t","venue":null,"work_id":"0a220bdd-a958-4ae9-80ac-09aa7aee7f86","year":null},"citing_paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T14:19:55.465965Z"},"links":{"citing_paper":"/paper/2505.19488"},"observation_digest":"sha256:17fea634db0bceee755ecaeceab0a54d2c91f8e0573a92fee71218fc994f50be","observation_id":"66cd9124-6ece-4b52-91dd-e926052f15f1","resolution":{"observed_at":"2026-08-07T14:19:56.312695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.19488","last_updated":"2025-05-26T04:15:38Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T01:31:51.166510Z","submitted_at":"2025-05-26T04:15:38Z","title":"Understanding Transformer from the Perspective of Associative Memory"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":1,"verified_fuzzy":18},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 17 inbound Pith citation observations for arXiv:2505.19488."}