{"as_of":"2026-08-10T08:37:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b010eb2fe8e58cbe5ccf539e5117a5225b1e08622dfce92126fa9219b7ca17fa","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T20:12:20.003565Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T04:14:18.393326Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05164","snapshot_observed_at":"2026-08-03T04:14:18.393326Z","title":", Bietti, A","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05725","last_updated":"2026-06-03T16:40:26Z","snapshot_observed_at":"2026-08-03T04:14:09.367178Z","submitted_at":"2026-02-05T14:49:40Z","title":"Muon in Associative Memory Learning: Training Dynamics and Scaling Laws","version":3},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-03T04:14:18.393326Z"},"links":{"cited_paper":"/paper/2502.05164","citing_paper":"/paper/2602.05725"},"observation_digest":"sha256:9f0ac298600e229b8125c410d82b02444de64863b18c76603d9902cc4c208374","observation_id":"674a9e47-8fd6-4ebc-9cee-50ab974d3e94","resolution":{"observed_at":"2026-08-03T04:14:18.393326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.05164/citation-record","integrity":"/paper/2502.05164/integrity","json":"/paper/2502.05164/citation-record.json","paper":"/paper/2502.05164"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.721191Z","title":"Transformers learn to implement preconditioned gradient descent for in-context learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.721191Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:1e5cd225245248d7b7d99cf859f89525633f4df02c83bc2207c55a0312c46229","observation_id":"91004943-45a4-491a-8a50-1b85437a1f2e","resolution":{"observed_at":"2026-08-08T20:12:19.721191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.809278Z","title":"What learning algorithm is in-context learning? investigations with linear models","venue":null,"work_id":"f12df134-135c-483d-ba57-861acae0ee9c","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.727883Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:d5553b7f9dde6de31a7578ab6601627366b4624472e0660744651f02232233d6","observation_id":"2877f911-2b69-4805-8814-c2bb08c1b3f4","resolution":{"observed_at":"2026-08-08T20:12:20.814429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.15571","last_updated":"2023-03-09T15:18:40Z","snapshot_observed_at":"2026-08-08T19:32:58.342274Z","submitted_at":"2022-09-30T16:30:31Z","title":"Building Normalizing Flows with Stochastic Interpolants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.15571","snapshot_observed_at":"2026-08-08T20:12:19.733758Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.733758Z"},"links":{"cited_paper":"/paper/2209.15571","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:650edfcabbd010cc35f612b9504ac021b2a52098089c35225a5fdf454634c16a","observation_id":"b6c0baf3-5fc8-421b-aabb-2ce801e5d6a2","resolution":{"observed_at":"2026-08-08T20:12:19.733758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.791968Z","title":"Learning patterns and pattern sequences by self-organizing nets of threshold elements","venue":null,"work_id":"693395af-3600-420e-811f-cbdf7edd4ecd","year":1972},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.740649Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:0f47840a4292466311d4c0adc601bd4c92dd3c0117d0e0fb2646e99b3216ccda","observation_id":"6ae27e66-9b13-48ee-ab10-839fb02b8660","resolution":{"observed_at":"2026-08-08T20:12:20.798226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.745757Z","title":"In search of dispersed memories: Generative diffusion models are associative memory networks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.745757Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:18bab546b70d91ddaa37f0c5a4c778f52362c058712578a1fdc751c7e1bb60cb","observation_id":"4d8c0977-5f99-4bdb-84ac-672326f4985e","resolution":{"observed_at":"2026-08-08T20:12:19.745757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.751221Z","title":"J., Gutfreund, H., and Sompolinsky, H","venue":null,"work_id":null,"year":1985},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.751221Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:644ce1f5c3359ad7e1db080469f9033b2e28e08287bb5ea148fe9d041bbbd0b4","observation_id":"0224f3c6-1bb8-4b94-894e-3915ce98ea3b","resolution":{"observed_at":"2026-08-08T20:12:19.751221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.773730Z","title":"M., Castillo, I","venue":null,"work_id":"7d476920-1462-4e50-a427-ed5fae815db8","year":2003},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.759533Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:ecd40047f329bf130d410c2089ae7500afedc1e14c35d6f276bf027463a15477","observation_id":"def250aa-c5fc-40f6-a925-0ce3c81078ee","resolution":{"observed_at":"2026-08-08T20:12:20.779421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.765137Z","title":"D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.765137Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:07621ad09ac120349c4822fa63309d3c30f4a27481c5090991b11adce6a8ab44","observation_id":"8cf6c370-5ab8-4e73-aa2d-67bb43ad5a0d","resolution":{"observed_at":"2026-08-08T20:12:19.765137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.744538Z","title":"Skyformer: Remodel self-attention with gaussian kernel and nyström method","venue":null,"work_id":"ec0fee7d-c643-4bd0-8652-81a433a35264","year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.771392Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:942764dd7a5d4509c91fdeb027b4d53fa947f90392af0dddde91010cdadf8603","observation_id":"c513ae5b-17af-4014-9276-548b8989ca39","resolution":{"observed_at":"2026-08-08T20:12:20.750265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.776643Z","title":"M., Likhosherstov, V., Dohan, D., Song, X., Gane, A., Sarlos, T., Hawkins, P., Davis, J","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.776643Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:57513da269b9926880eba4a780ca569828988ad8bde0d5b4590eaae5a73bf8e5","observation_id":"3ff74433-c413-4b49-a643-c5be97014610","resolution":{"observed_at":"2026-08-08T20:12:19.776643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10559","last_updated":"2023-05-15T11:45:12Z","snapshot_observed_at":"2026-08-09T08:03:27.441735Z","submitted_at":"2022-12-20T18:58:48Z","title":"Why Can GPT Learn In-Context? Language Models Implicitly Perform Gradient Descent as Meta-Optimizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10559","snapshot_observed_at":"2026-08-08T20:12:19.783838Z","title":"Why can gpt learn in-context? language models implicitly perform gradient descent as meta-optimizers, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.783838Z"},"links":{"cited_paper":"/paper/2212.10559","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:7cb8d60fee5ff867279640f4f471fc4301a6c7cc076a9996324017bb9f2a65f1","observation_id":"13b5fa6c-41bf-43a1-8b11-d7a44164c3ca","resolution":{"observed_at":"2026-08-08T20:12:19.783838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.716348Z","title":"On a model of associative memory with huge storage capacity","venue":null,"work_id":"909081f3-aa5a-4d0c-9d71-a3acf5414abd","year":2017},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.789943Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:eae0103679ead435be706494ba73b017a0e84928c616b149c1a55d057805097e","observation_id":"3264bd70-f400-4df2-9ce7-81f0e6ae447b","resolution":{"observed_at":"2026-08-08T20:12:20.722080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.698807Z","title":"Bert: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":"9ee3e53e-7dd7-462b-bf49-b753b37b713f","year":2019},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.795289Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:b6d7d96130180df114a4183621ec568cb8379f02b6d818a9e54e272bf16b4e98","observation_id":"aa2f8aff-c2cd-4c57-aa5a-25f824b86ae7","resolution":{"observed_at":"2026-08-08T20:12:20.704500Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.801432Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.801432Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:619cd17f371663af7ca097b72c2ae45736aea53296ceb873806ac448d4ea0e89","observation_id":"b61d8712-34c0-433e-81c8-e426144f182d","resolution":{"observed_at":"2026-08-08T20:12:19.801432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.670078Z","title":null,"venue":null,"work_id":"7f863e7f-bdf5-4cae-b9da-641d63c4a307","year":1993},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.807071Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:bdd77deff5289aa94ab434dddd84018734155d8cc257dc79c49f35cb6126c3d7","observation_id":"baa23eab-c0b2-4263-b57b-9fb508fad702","resolution":{"observed_at":"2026-08-08T20:12:20.675115Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.01066","last_updated":"2023-08-11T19:27:58Z","snapshot_observed_at":"2026-08-09T18:53:39.110458Z","submitted_at":"2022-08-01T18:01:40Z","title":"What Can Transformers Learn In-Context? A Case Study of Simple Function Classes","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.01066","snapshot_observed_at":"2026-08-08T20:12:19.813203Z","title":"S., and Valiant, G","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.813203Z"},"links":{"cited_paper":"/paper/2208.01066","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:71cc8d53c6991fbd479a8126c92f749557e5216a7f76a245b0966ba8b9d0038f","observation_id":"e311b2af-b77d-4843-8f8d-9661f34e8b51","resolution":{"observed_at":"2026-08-08T20:12:19.813203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.654105Z","title":"Sampling with flows, diffusion, and autoregressive neural networks from a spin-glass perspective","venue":null,"work_id":"6ae4c702-b8a8-42d5-b12f-332de86fb8fd","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.819532Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:8366ff700194c0b166d9ef33082fae5f5f4b36d5063ddf3842cc0042d2eeccb9","observation_id":"769bf4ae-bd6c-4187-9632-39bdccf7310e","resolution":{"observed_at":"2026-08-08T20:12:20.659317Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.637898Z","title":null,"venue":null,"work_id":"1fe8badf-8e8a-4558-9442-c8cc23788dc5","year":2007},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.825224Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:8c4ab22214e791f3676cbbad6b619005fd01ccaa9e030f29a0ec409c2bc50e3e","observation_id":"cfc00335-dc6e-407c-bf5c-92308aee9f33","resolution":{"observed_at":"2026-08-08T20:12:20.642930Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.830090Z","title":"Probability inequalities for sums of bounded random variables","venue":null,"work_id":null,"year":1994},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.830090Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:3fc2717a490c3e24f78d9b0dd5d77253b8aff3cb87fe4ec17f34e45c42929dcd","observation_id":"f07c8e26-3afb-48ab-b07d-3139de192457","resolution":{"observed_at":"2026-08-08T20:12:19.830090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.611893Z","title":"H., Zaki, M., and Krotov, D","venue":null,"work_id":"fa4cea02-89fa-4a6e-8476-20d15798983b","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.834946Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:2ab6e37d366184330bf371c1759b1860e62f9ded4242f96052baa952be556f40","observation_id":"539c970a-ebe3-4f4e-bb01-a3b8df0df8fd","resolution":{"observed_at":"2026-08-08T20:12:20.616978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.595053Z","title":"H., Zaki, M","venue":null,"work_id":"7497231d-24f5-437d-8a99-19a28feff0ad","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.840519Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:be759feda94cb43ea658a1e533389a5d8d9587480b405d24e81834faa0e71f32","observation_id":"51534cf1-08d0-4c05-b897-13622a9d4f7a","resolution":{"observed_at":"2026-08-08T20:12:20.600289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.578639Z","title":"H., Strobelt, H., Ram, P., and Krotov, D","venue":null,"work_id":"ea84682f-4a89-4279-8c87-2fbd4fd8e636","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.845924Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:8b017379ebe54dad45821ca507cbe08ce116b349cfeecfd016e54f72ddea32b7","observation_id":"a510701a-a36a-413f-8e47-a9b94bf78bca","resolution":{"observed_at":"2026-08-08T20:12:20.583798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16750","last_updated":"2024-05-28T11:46:33Z","snapshot_observed_at":"2026-08-10T08:01:29.478810Z","submitted_at":"2023-09-28T17:57:09Z","title":"Memory in Plain Sight: Surveying the Uncanny Resemblances of Associative Memories and Diffusion Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16750","snapshot_observed_at":"2026-08-08T20:12:19.851996Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.851996Z"},"links":{"cited_paper":"/paper/2309.16750","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:f336c8870924488ab75b41d61dace406b1915040a306720a1e27935dfccea304","observation_id":"a11133fd-e21e-40ec-890a-9eb59bec96b3","resolution":{"observed_at":"2026-08-08T20:12:19.851996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.858163Z","title":null,"venue":null,"work_id":null,"year":1982},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.858163Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:fe1ec8ef1652717a57b94813a10bf67cea70fa1cf232f69889f4977657b301c2","observation_id":"092e970c-c195-478a-a96e-0298498cefef","resolution":{"observed_at":"2026-08-08T20:12:19.858163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.561442Z","title":"Y.-C., Yang, D., Wu, D., Xu, C., Chen, B.-Y., and Liu, H","venue":null,"work_id":"39429644-ef26-4b60-af75-5d90b0b10559","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.863027Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:3ae2615308f4747b25ffa4370089b4e40739f611b82e2a3b0cce39efdc85e3f2","observation_id":"fdb177ab-34ce-4722-9b26-74adebe7242b","resolution":{"observed_at":"2026-08-08T20:12:20.567487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.868578Z","title":"Transformers are rnns: fast autoregressive transformers with linear attention","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.868578Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:99ef27389a86863f495db6ebfb974d235ed72c093720f7b5400578e3c4ef2a5a","observation_id":"7029d907-01e2-4041-a9c6-4428812673cd","resolution":{"observed_at":"2026-08-08T20:12:19.868578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.534504Z","title":"A new frontier for hopfield networks","venue":null,"work_id":"e428d6ac-a7db-4d69-b370-c07b80ac50b7","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.873476Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:6b5114e5b066806ce042af0d7cc38442019196ad3137eb760b42b4eeaaa13571","observation_id":"054cbaa4-8b89-43af-96d7-49798184ca88","resolution":{"observed_at":"2026-08-08T20:12:20.540752Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.518730Z","title":"and Hopfield, J","venue":null,"work_id":"b8cbff2d-f6a1-4a03-9763-1b6f1e150647","year":2016},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.879407Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:979f7655b024432fbc621be9db2b8d0330a0536d755348fda0586709df2cfeca","observation_id":"01f9dcb1-d452-4d48-b8c8-52095f87deae","resolution":{"observed_at":"2026-08-08T20:12:20.523648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.500646Z","title":"and Hopfield, J","venue":null,"work_id":"52847c59-8d06-4aaa-a58c-6518b055cf6e","year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.885552Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:35fec35fa790ce5409f1b3707a71dae3178620dd48eac4316ec47a2f52213468","observation_id":"09da571a-f368-4b7c-8335-e9094e7ca383","resolution":{"observed_at":"2026-08-08T20:12:20.506949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.484517Z","title":null,"venue":null,"work_id":"dde23295-da6d-4004-89df-55905d200aee","year":1974},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.890725Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:6fd27d94212410587449d2f1e2844be900bd98c8d81d9f9e7164e4f311b839b4","observation_id":"90a63058-3566-4740-9277-99d52c383bbc","resolution":{"observed_at":"2026-08-08T20:12:20.489411Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.468102Z","title":"Probability theory i","venue":null,"work_id":"7500c4a0-bac3-4ec3-beac-04192e9fd037","year":1977},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.895696Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:edfc5882b43053b95ca2aebbf2fdda8dd21550025568c846f4e8e2ef41ba1230","observation_id":"2aeac261-eae3-4d6d-8baa-3f6302b93b13","resolution":{"observed_at":"2026-08-08T20:12:20.473411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.900776Z","title":"and M\\'ezard, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.900776Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:cf745dce40d9916cad760f9fe62753aa8aae9dfd07f6b1d4bbf85f4a345ce70f","observation_id":"ac48d440-57a3-4191-88df-e5a134d4d0ba","resolution":{"observed_at":"2026-08-08T20:12:19.900776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.450686Z","title":"Universal hopfield networks: A general framework for single-shot associative memory models","venue":null,"work_id":"98d0c99b-5f3b-4ce3-8c74-037ed822ca40","year":2022},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.905910Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:467c2f6c4a34bd06c0ba88fe1945453281e6b5b112e9cfaa0f4de0caa6cabac3","observation_id":"c0fdd48e-f83b-4340-9898-92dd80e36e0d","resolution":{"observed_at":"2026-08-08T20:12:20.455811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.434304Z","title":"Associatron-a model of associative memory","venue":null,"work_id":"7791e6df-961b-4edf-b603-ebdfd9577f7a","year":1972},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.910946Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:ddbb0c090d0f12b8747b2e2f59e3516c4a51b920c76533d4b893e6ebe0da990d","observation_id":"970f2bd5-9310-4143-9bbe-c3408fa135b5","resolution":{"observed_at":"2026-08-08T20:12:20.439551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.418170Z","title":"J., Ambrogioni, L., and Krotov, D","venue":null,"work_id":"7055f975-b7da-4b18-b4c5-266d33711551","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.916303Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:14f01d06f625c00e0415eb07bcdbdef10737bc5b66e8b51295cec72fb20019ce","observation_id":"db9ce9da-d70e-48da-aa4c-2022d15e0e90","resolution":{"observed_at":"2026-08-08T20:12:20.423397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.402506Z","title":"P., Kopp, M","venue":null,"work_id":"d091144f-3a68-404f-9d4b-7d32bbdeddb8","year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.922685Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:d15a5a70f02aa73977bb619997cd5cb8b320ee2fae5f2064934518a24fb63ca1","observation_id":"2e5c9f53-790f-4349-956c-26e1ce643618","resolution":{"observed_at":"2026-08-08T20:12:20.407286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.382432Z","title":"The mechanistic basis of data dependence and abrupt learning in an in-context classification task","venue":null,"work_id":"fd3210e7-2fba-4761-bb4d-f26f447066f0","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.927878Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:83e90afa2f5a3e8da30f809339f599e12782b48639c157eade159cd1225d7134","observation_id":"5eb13efe-e8cc-40f4-b31c-d6163978bcf3","resolution":{"observed_at":"2026-08-08T20:12:20.389476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19244","last_updated":"2023-10-30T03:22:54Z","snapshot_observed_at":"2026-08-10T01:05:13.838009Z","submitted_at":"2023-10-30T03:22:54Z","title":"High-Dimensional Statistics","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19244","snapshot_observed_at":"2026-08-08T20:12:19.933271Z","title":"and H \\\"u tter, J.-C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.933271Z"},"links":{"cited_paper":"/paper/2310.19244","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:c730a78decc157a46af41e5a3c04d677f4f422ba0621162841852180b484c56f","observation_id":"9582f500-aae4-475c-9a3a-8756dc0845f1","resolution":{"observed_at":"2026-08-08T20:12:19.933271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.363885Z","title":null,"venue":null,"work_id":"0afc0613-9e62-4f80-8c92-f463957e035b","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.939249Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:cc4ba7a10ff497f39d3faa272ffb7dea866ac09a094f42e88b34f6eddd439704","observation_id":"325c1420-7247-4f74-85df-1c797b5220f4","resolution":{"observed_at":"2026-08-08T20:12:20.369749Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.346237Z","title":null,"venue":null,"work_id":"23aadbf7-bd1f-4203-b871-3214ecbcad13","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.945246Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:3fb8e10d9b26746e13cc476be9f0b58813cf1c9f984abc6b8ad1094a3c0e4e61","observation_id":"38837b5d-7cd3-4004-81c9-46d27159bcdc","resolution":{"observed_at":"2026-08-08T20:12:20.351319Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.328874Z","title":"and Zilman, A","venue":null,"work_id":"09e86186-e62e-434e-888b-06f580473d08","year":2021},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.952676Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:73022f628fb3bc037c144f9804cad42327b9d3f0271d2c4a0e675b9c7bdf4c76","observation_id":"bbe52dc3-773d-4cc9-a00a-92713f5de50b","resolution":{"observed_at":"2026-08-08T20:12:20.334330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-08T20:12:19.957739Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.957739Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:bdd765ae3a9e51b990dc490758708d303d6573b1f960f9f3e783e6382a9d8331","observation_id":"821c45df-7062-4b54-927d-720e3278b637","resolution":{"observed_at":"2026-08-08T20:12:19.957739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.312161Z","title":"and Kolter, J","venue":null,"work_id":"9ca4129e-b5ef-4cb2-aa65-74a5db0c699b","year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.963489Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:c23922e31f25ab6e2d9d796b6e34704d48fb971991b17b23bc5b50b909c145a7","observation_id":"625ed1f4-5df5-48c7-ad58-e6f618dbf024","resolution":{"observed_at":"2026-08-08T20:12:20.317383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.293935Z","title":"N., Łukasz Kaiser, and Polosukhin, I","venue":null,"work_id":"a1f198bf-61a9-4418-ac43-2e9f1055f97a","year":2017},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.969519Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:7912643be78cfa5bdc5079f4d08bd1352de315dd8382b404407a8b5e26b0b8fd","observation_id":"38841fad-e635-4bd1-8234-6821994bd7d2","resolution":{"observed_at":"2026-08-08T20:12:20.299496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.974016Z","title":"Transformers learn in-context by gradient descent","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.974016Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:25b27c486c24832dea52d47bfc40762742c73bab1a1abed31b4477462127dff1","observation_id":"f203ca04-c781-45fb-b34a-0e80a415dbc0","resolution":{"observed_at":"2026-08-08T20:12:19.974016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.267847Z","title":"Y.-C., Hsiao, T.-Y., and Liu, H","venue":null,"work_id":"a995ed0a-5515-4f1a-a82a-73f399703acb","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.978627Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:df364cec5ce838decfbc25b814865d50365974452fb662881bbda59575f44734","observation_id":"5f3b6b87-d9b7-470c-b9f5-60749dc1f2c0","resolution":{"observed_at":"2026-08-08T20:12:20.272991Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.17346","last_updated":"2023-12-28T20:26:23Z","snapshot_observed_at":"2026-08-09T03:14:51.041161Z","submitted_at":"2023-12-28T20:26:23Z","title":"STanHop: Sparse Tandem Hopfield Model for Memory-Enhanced Time Series Prediction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.17346","snapshot_observed_at":"2026-08-08T20:12:19.983057Z","title":"Y.-C., Li, W., Chen, B.-Y., and Liu, H","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.983057Z"},"links":{"cited_paper":"/paper/2312.17346","citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:25d7d2dbc547a229db3f653a2f1b1cb782e38fea07e2f802e91475f117d386d1","observation_id":"17a42cd6-5863-467e-a1a3-180683bd7195","resolution":{"observed_at":"2026-08-08T20:12:19.983057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.248680Z","title":null,"venue":null,"work_id":"4319dd9d-0c1b-4d09-a19d-88c87ad4fad9","year":2024},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.987968Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:f0e9f736d1809d3d7bec1710b26f4f33309b2e68a5d0378fccfdc3bbee0f143a","observation_id":"7f0d2761-8324-43c4-bbc1-f32650ad520a","resolution":{"observed_at":"2026-08-08T20:12:20.255153Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.992948Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.992948Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:eddba6602a3ba914a1529c27fa7c0684b24d3369519736ee61f583144fda0a8c","observation_id":"5754da27-3fab-401a-9095-c25a02eca7e7","resolution":{"observed_at":"2026-08-08T20:12:19.992948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:19.998349Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:19.998349Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:3911ac691c7df791fafb7e5f8b9db44c9fe4f07dfe690062e20fcaba0a3c83ce","observation_id":"e2831b3b-3e81-4625-9336-08314362cf70","resolution":{"observed_at":"2026-08-08T20:12:19.998349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:12:20.003565Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-08T20:12:20.003565Z"},"links":{"citing_paper":"/paper/2502.05164"},"observation_digest":"sha256:8ed3b54d2d5f651dfe26a2730037267ad70268ccd40bffcd77ed09cd10f49805","observation_id":"55189220-1631-4dfa-9c08-b2061ffb4597","resolution":{"observed_at":"2026-08-08T20:12:20.003565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.05164","last_updated":"2025-06-06T05:00:45Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T00:03:05.101995Z","submitted_at":"2025-02-07T18:48:25Z","title":"In-context denoising with one-layer transformers: connections between attention and associative memory retrieval"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":27,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 1 inbound Pith citation observation for arXiv:2502.05164."}