{"as_of":"2026-08-09T18:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:906a515ecf552e46890df82dfc7ab82c43c4f3122f44f806f84732108a412ab5","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:57:50.562642Z","state":"measured"},{"denominator":89,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":89,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.01414/citation-record","integrity":"/paper/2507.01414/integrity","json":"/paper/2507.01414/citation-record.json","paper":"/paper/2507.01414"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.12973","last_updated":"2024-01-30T18:59:34Z","snapshot_observed_at":"2026-07-06T17:19:31.847730Z","submitted_at":"2024-01-23T18:59:21Z","title":"In-Context Language Learning: Architectures and Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.12973","snapshot_observed_at":"2026-08-06T20:57:47.325633Z","title":"In-context language learning: Architectures and algorithms.arXiv preprint arXiv:2401.12973, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.325633Z"},"links":{"cited_paper":"/paper/2401.12973","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:91d19cdeb0b5a6cb8a9634e695bae13652e231ca612e1a0ef0fb49ce08a17441","observation_id":"5d63445b-9904-4aa1-9fcd-352603307b18","resolution":{"observed_at":"2026-08-06T20:57:47.325633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.391018Z","title":"Lepori, Jack Merullo, and Ellie Pavlick","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.391018Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:93ba87b67ad2848d23e5f1b875e23a72c9714f736868d8057e4af0f17be262e9","observation_id":"68b19d44-e34d-4448-a1b7-0f8fc71f69b8","resolution":{"observed_at":"2026-08-06T20:57:47.391018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04927","last_updated":"2023-12-08T09:44:25Z","snapshot_observed_at":"2026-08-05T07:11:50.808005Z","submitted_at":"2023-12-08T09:44:25Z","title":"Zoology: Measuring and Improving Recall in Efficient Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04927","snapshot_observed_at":"2026-08-06T20:57:47.494240Z","title":"Zoology: Measuring and improving recall in efficient language models.arXiv preprint arXiv:2312.04927, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.494240Z"},"links":{"cited_paper":"/paper/2312.04927","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:0a844ba7e111033e985f971db140018c6722383a58ebdb71c525d6432a434e52","observation_id":"c6d43768-d201-4cbb-9be7-612c124966d4","resolution":{"observed_at":"2026-08-06T20:57:47.494240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.557374Z","title":"Copernicus, New York, NY , USA, 1996","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.557374Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7019ef526cb10487f34e6a9c289fce9dae9875d48af396afe1fcdc7d92efb6dd","observation_id":"f5ac2633-72d9-4809-ba23-b5ac0b410e82","resolution":{"observed_at":"2026-08-06T20:57:47.557374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.637617Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.Transactions on Machine Learning Research, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.637617Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:74f15a33515df3b29a1cdbf76046339126a3b7b343f60924575d5d9e57690335","observation_id":"207607ea-25dc-4828-8ae7-4bb8ce2d546f","resolution":{"observed_at":"2026-08-06T20:57:47.637617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.729232Z","title":"Finding transformer circuits with edge pruning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.729232Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e1137eda4f75d718ea063387accd4cb0787fefbf4edb3801d39fa3f1d904dbcc","observation_id":"032bf900-263b-4253-9a51-6e01fb661c56","resolution":{"observed_at":"2026-08-06T20:57:47.729232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.830622Z","title":"Language models are few-shot learners.Advances in neural information processing systems, 33:1877–1901, 2020","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.830622Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:df73cb5613f1ef55b80d6d80b7d3cf605a247fa94d4e6efe530abb1ce9ab5ed0","observation_id":"78db4f16-32ca-47de-b375-2d4c0c16fbb0","resolution":{"observed_at":"2026-08-06T20:57:47.830622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.922775Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.922775Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:be1073ca98b9752a7daa2f34cf6113d36d9fa59058af5597e64d3588933f54cf","observation_id":"db2150c0-0f3e-41b5-8df8-7b6ffbd542bb","resolution":{"observed_at":"2026-08-06T20:57:47.922775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.990675Z","title":"Toward understanding in-context vs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.990675Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:dbd707258e1e93fde67bc8ebe8b032fe216770ad3fd27dfa644162ac9092b9b6","observation_id":"8f2873f2-e3ae-4c09-a79f-3e64fb34b23e","resolution":{"observed_at":"2026-08-06T20:57:47.990675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.058830Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.058830Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:39e20ecfa225517809ec044ca0a9a937f78774598001d9e06fdfe82a072a486a","observation_id":"2b8c0232-67ae-42da-9f40-579c5dce449a","resolution":{"observed_at":"2026-08-06T20:57:48.058830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.125248Z","title":"Sudden drops in the loss: Syntax acquisition, phase transitions, and simplicity bias in MLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.125248Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:69fc87600fec74c0468d1c5e0264cdbe759e75567da535b6fe4f0270c32178bb","observation_id":"155da576-b9f3-48bd-9e5f-055be0e4a8a8","resolution":{"observed_at":"2026-08-06T20:57:48.125248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.684738Z","title":"Quantifying semantic emergence in language models, 2024","venue":null,"work_id":"ba5c0422-dce1-426c-92e2-6b89604b982c","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.199456Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:5eccd49799c87b79558e544ea5701155af1585b90800b4fedab0e1e4b90ba687","observation_id":"b051e9bc-f5c2-4236-8f43-17bc5f1db24e","resolution":{"observed_at":"2026-08-06T20:57:51.689610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.670424Z","title":"Dynamical versus bayesian phase transitions in a toy model of superposition, 2023","venue":null,"work_id":"7514f4fd-bde6-46bd-ac3e-83895712c48e","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.285810Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:da66b4faf398c35c345919c8bcb5dd7920d08f990f97db1335b65198326e3119","observation_id":"bc162cb3-65db-437b-a29f-d57d543a9333","resolution":{"observed_at":"2026-08-06T20:57:51.675192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.06173","last_updated":"2023-03-10T19:16:53Z","snapshot_observed_at":"2026-07-06T15:01:30.019542Z","submitted_at":"2023-03-10T19:16:53Z","title":"Unifying Grokking and Double Descent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.06173","snapshot_observed_at":"2026-08-06T20:57:48.356869Z","title":"Unifying grokking and double descent","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.356869Z"},"links":{"cited_paper":"/paper/2303.06173","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:2e63b5929fcbd890d122277d4614ecffd390055fa0c305fd65dafa0c43d6ce88","observation_id":"9ace98a6-1810-4e4e-b52f-24edb86128f2","resolution":{"observed_at":"2026-08-06T20:57:48.356869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.655863Z","title":"Can transformers learn optimal filtering for unknown systems?IEEE Control Systems Letters, 7:3525–3530, 2023","venue":null,"work_id":"0bac690a-ba8f-4ce3-a866-d7d60046824b","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.421421Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e19c4e90c5db9dc5b6c1da816563ec13ec5752e0da7ddadca4184263370cf4af","observation_id":"36919971-e03f-4af6-87ec-6f6dbf4a9630","resolution":{"observed_at":"2026-08-06T20:57:51.660506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.640289Z","title":"Understanding emergent abilities of language models from the loss perspective, 2025","venue":null,"work_id":"6ed6f7e8-416b-4892-9cc2-19189c6dd57d","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.489563Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:769dfaeb8a1219de245799db7e669158d66be83ae5551f49a129030031cba7f3","observation_id":"d5873c6d-c0ee-4533-a4bd-ff4e6cce04e3","resolution":{"observed_at":"2026-08-06T20:57:51.645956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11004","last_updated":"2024-02-16T18:28:36Z","snapshot_observed_at":"2026-08-03T19:49:17.673100Z","submitted_at":"2024-02-16T18:28:36Z","title":"The Evolution of Statistical Induction Heads: In-Context Learning Markov Chains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11004","snapshot_observed_at":"2026-08-06T20:57:48.557237Z","title":"The evolution of statistical induction heads: In-context learning markov chains.arXiv preprint arXiv:2402.11004, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.557237Z"},"links":{"cited_paper":"/paper/2402.11004","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7de8647a6582488898c2bb0dc80e783c3824757a9210bbd1a6818765812b708b","observation_id":"6f0fa652-2942-45fe-a97e-2280ce290262","resolution":{"observed_at":"2026-08-06T20:57:48.557237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.634788Z","title":"A mathematical framework for transformer circuits.Transformer Circuits Thread,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.634788Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d29aae497a5c0debf1b8522658a79161fbe26af48989a957f7094abfe069deef","observation_id":"ee229cd8-fb24-4b55-9fe6-b5c160e377e9","resolution":{"observed_at":"2026-08-06T20:57:48.634788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.606042Z","title":"Predictability and surprise in large generative models","venue":null,"work_id":"902687e1-e391-4d3a-acab-5d032c04d6d3","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.802653Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b55f4c3a8c834ce83fa67e34985a826907a9027ffc9f429e8f1eeb47e2e754fb","observation_id":"d7461706-7e55-4ca8-9e43-a9e4ee485eb2","resolution":{"observed_at":"2026-08-06T20:57:51.611107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.874830Z","title":"What can transformers learn in-context? a case study of simple function classes.Advances in Neural Information Processing Systems, 35:30583–30598, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.874830Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:6496d43d6496b61a302e0157810e6a8fd7f87a9baf898d1a2e96fd8231071a22","observation_id":"f523d299-edc4-48a9-ade2-efcce5a27bf7","resolution":{"observed_at":"2026-08-06T20:57:48.874830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00838","last_updated":"2024-06-07T21:59:52Z","snapshot_observed_at":"2026-07-06T17:23:51.547578Z","submitted_at":"2024-02-01T18:28:55Z","title":"OLMo: Accelerating the Science of Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00838","snapshot_observed_at":"2026-08-06T20:57:48.954246Z","title":"Olmo: Accelerating the science of language models.arXiv preprint arXiv:2402.00838, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.954246Z"},"links":{"cited_paper":"/paper/2402.00838","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:423e931f6de4466cf2cc054de854a8cfe6d3d561a4f2101db71094c44d3427c5","observation_id":"33e27c69-de1b-4d9d-81d4-70126a5b76c8","resolution":{"observed_at":"2026-08-06T20:57:48.954246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02679","last_updated":"2023-01-06T19:00:01Z","snapshot_observed_at":"2026-08-06T08:20:11.326697Z","submitted_at":"2023-01-06T19:00:01Z","title":"Grokking modular arithmetic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02679","snapshot_observed_at":"2026-08-06T20:57:49.031244Z","title":"Grokking modular arithmetic.arXiv preprint arXiv:2301.02679, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.031244Z"},"links":{"cited_paper":"/paper/2301.02679","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:55db18f7ca79c3512578d64368616cb2337d13729feffca93b379cd7bd1c4f34","observation_id":"c11acdc7-b915-4c30-bbee-d21ed541daf4","resolution":{"observed_at":"2026-08-06T20:57:49.031244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.580758Z","title":"Loss landscape degeneracy drives stagewise development in transformers, 2025","venue":null,"work_id":"af880b6b-834b-4016-ab86-6a74d41f26c2","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.103661Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:4068b515dad2ad417ea55c6431cac689c1ce657de3ff056e36db951271d156d5","observation_id":"7d6cc3ad-fdba-435e-bbe0-8ec343865c1a","resolution":{"observed_at":"2026-08-06T20:57:51.586092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.187011Z","title":"Neural networks and physical systems with emergent collective computational abilities.Proceedings of the national academy of sciences, 79(8):2554–2558, 1982","venue":null,"work_id":null,"year":1982},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.187011Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d9fb02a96ac918cc69875ddfa4e413f840c0decaf5b160f1509e55afc061b9f9","observation_id":"fd601a35-8b4c-4b53-a661-43ef27000f66","resolution":{"observed_at":"2026-08-06T20:57:49.187011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.554727Z","title":"Task descriptors help transformers learn linear models in-context","venue":null,"work_id":"e4fcb9cf-c98c-4c47-86b7-8cf3c285cd6c","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.284731Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:27139646fb6ac529bee54e0ba60b4027a685e4e3456abd8543b3c54b8a11c8cf","observation_id":"5fbf382b-11c6-4e9f-bcfa-8a1bcefa9927","resolution":{"observed_at":"2026-08-06T20:57:51.560528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.540443Z","title":"Deep networks always grok and here is why","venue":null,"work_id":"eca72e53-ccbf-47c1-a044-19b1f8d95ff7","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.292777Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ab6902af2589d698f1d161dc25373d638a016dcda456447c9bad374e348eede3","observation_id":"7ff2b510-b71f-43d7-94ff-6aa2798a8dae","resolution":{"observed_at":"2026-08-06T20:57:51.544878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.525797Z","title":null,"venue":null,"work_id":"cb7dd801-a1c4-4aae-aaaf-9cd75393edab","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.386295Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:07aeacf004067d9e4d38b21a42151f23e034524f120b9512443cd6c4851f5706","observation_id":"67e4b03f-617a-4dc8-b0a4-ce4bd50f7f8e","resolution":{"observed_at":"2026-08-06T20:57:51.530373Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.510425Z","title":"Grokking as the transition from lazy to rich training dynamics","venue":null,"work_id":"69de77e9-f44a-4018-bd5f-f64898ccde0e","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.521069Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:3ecdf7c097e2caeab8528dd37e7c1e3997516d99c3bb040e52bf5c4849ae0cbb","observation_id":"5780ac73-d04a-4e0c-b3b4-8316f3714c72","resolution":{"observed_at":"2026-08-06T20:57:51.515532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.494514Z","title":null,"venue":null,"work_id":"985bb60a-a72b-4228-9901-5c2d9e174dbe","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.640425Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c57883cc82411198b8eb7ba9afb8ab4b7054f1d973f7688173c217d2cd4cc577","observation_id":"1aa3af2f-3150-4b30-a59a-66594e8a73f9","resolution":{"observed_at":"2026-08-06T20:57:51.499994Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.480240Z","title":"The local learning coefficient: A singularity-aware complexity measure, 2024","venue":null,"work_id":"92a18f35-c9cb-4658-9448-8c798d51dc32","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.772445Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:38c26f083c7a6da6d1c278efb8d7f875b16a070ecbe810b6d40a3660b29323ec","observation_id":"64c6aff3-b30d-44fe-abd8-6fadb4733a06","resolution":{"observed_at":"2026-08-06T20:57:51.484652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14578","last_updated":"2024-10-28T04:10:40Z","snapshot_observed_at":"2026-07-06T18:18:39.093732Z","submitted_at":"2024-05-23T13:52:36Z","title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","version":5},"cited_work":{"arxiv_id":"2405.14578","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.14578","snapshot_observed_at":"2026-08-06T20:57:50.724970Z","title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","venue":"cs.LG","work_id":"289492d8-e0ad-4faa-b970-1cad93a5e2e1","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.881070Z"},"links":{"cited_paper":"/paper/2405.14578","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fddef3287cf038320e390f8c4d69e4dbdb3be3e0a25310759b0db53ddaa1a71a","observation_id":"cca60b8a-8980-4b77-abc3-e7d778eb3db3","resolution":{"observed_at":"2026-08-06T20:57:50.729788Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.465792Z","title":"Trans- formers as algorithms: Generalization and stability in in-context learning","venue":null,"work_id":"1b0ddba0-ca96-4a0b-8ee1-311e36417258","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.007600Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:f44673dd1c28277e7b825d2e085b5b826f158d51039d2218a04f108611537e92","observation_id":"f77a2e13-8aeb-4532-8a80-ea132a00d750","resolution":{"observed_at":"2026-08-06T20:57:51.470598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.451384Z","title":"Dual operating modes of in-context learning, 2024","venue":null,"work_id":"d9e85277-b706-4175-aad9-aaed57e046f3","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.124977Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:6d6c346b3beda6ebf283309e0e03dadfb1da41e8d09886fddb8e0ad1bc537295","observation_id":"99bfa400-d578-497d-bea8-ea8c3d4b4f5e","resolution":{"observed_at":"2026-08-06T20:57:51.456102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.437420Z","title":"Can transformers solve least squares to high precision? InICML 2024 Workshop on In-Context Learning, 2024","venue":null,"work_id":"ce89b72d-0fad-4e5f-9870-79547f729e94","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.145177Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:584b0733a277959eadadb3efe6f59c907daacf1197314372e7d0cb1bf742a887","observation_id":"f38d062f-5859-480d-8a95-86e6c878fafc","resolution":{"observed_at":"2026-08-06T20:57:51.441982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.201250Z","title":"Liu, Kevin Lin, John Hewitt, Ashwin Paranjape, Michele Bevilacqua, Fabio Petroni, and Percy Liang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.201250Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:6014b0a6a19450463767e82751a65cd583b381af83d5750e11f07b1e47fe59d9","observation_id":"7b828d0f-6eb3-461a-b0e5-a0942822f9e1","resolution":{"observed_at":"2026-08-06T20:57:50.201250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.414500Z","title":"Omnigrok: Grokking beyond algorithmic data","venue":null,"work_id":"4eb67290-712a-4fd0-adf5-0a234bf1a77e","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.316022Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ec58554f6290d36617ceac4d7594c665441e28b85d4b976c7e21e747daafed10","observation_id":"671c5f59-d0c5-47c4-ba43-fc4270979382","resolution":{"observed_at":"2026-08-06T20:57:51.418824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.327309Z","title":"Decoupled weight decay regularization, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.327309Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7d5ff9410ad309ad23808ac3761cbb0671cdeef0930086c9566da242b9fee0c7","observation_id":"2688f4e1-93c3-4562-ba48-a348e8927a75","resolution":{"observed_at":"2026-08-06T20:57:50.327309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.01809","last_updated":"2024-07-15T12:21:56Z","snapshot_observed_at":"2026-08-05T15:26:10.962792Z","submitted_at":"2023-09-04T20:54:11Z","title":"Are Emergent Abilities in Large Language Models just In-Context Learning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.01809","snapshot_observed_at":"2026-08-06T20:57:50.362140Z","title":"Are emergent abilities in large language models just in-context learning?arXiv preprint arXiv:2309.01809, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.362140Z"},"links":{"cited_paper":"/paper/2309.01809","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:1187c4d3a058c195213387768321ae111d6a42fa6fe156dd7f0e669e70ca539a","observation_id":"2e1679c7-fd4d-487e-9ea8-b2e1e801596d","resolution":{"observed_at":"2026-08-06T20:57:50.362140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.389966Z","title":"Dick, and Hidenori Tanaka","venue":null,"work_id":"3c082550-ce4c-452c-92b2-5706ab63e393","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.366585Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9a558605dcb1d8d4ec6f82cc6754cbce5fec613425b86a8a0b5232a3b733806a","observation_id":"0ddb44b2-94e3-4b9f-b5d3-04c049609a5a","resolution":{"observed_at":"2026-08-06T20:57:51.394536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18817","last_updated":"2024-04-02T05:43:18Z","snapshot_observed_at":"2026-08-09T15:53:01.595776Z","submitted_at":"2023-11-30T18:55:38Z","title":"Dichotomy of Early and Late Phase Implicit Biases Can Provably Induce Grokking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18817","snapshot_observed_at":"2026-08-06T20:57:50.370208Z","title":"Dichotomy of early and late phase implicit biases can provably induce grokking.arXiv preprint arXiv:2311.18817, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.370208Z"},"links":{"cited_paper":"/paper/2311.18817","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:590cacdd3c6296aeda07896eab95082b2427f7e30c4a7bed057b4859223b441a","observation_id":"af9cca3c-f0ec-4dc1-9f81-3963a84a4100","resolution":{"observed_at":"2026-08-06T20:57:50.370208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.375536Z","title":"A practical bayesian framework for backpropagation networks.Neural computation, 4(3):448–472, 1992","venue":null,"work_id":"2639d473-5147-40b2-87a4-d090a3bbb637","year":1992},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.374733Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:29aeeaf7a7790c21feab15e3d92ad7116b1975043de651f102d2a294fd4433ce","observation_id":"bb2a53ad-8667-469a-98be-b25573be7a5c","resolution":{"observed_at":"2026-08-06T20:57:51.380244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.361220Z","title":"Exact learning dynamics of in-context learning in linear transformers and its application to non-linear transformers, 2025","venue":null,"work_id":"0328eb25-74df-497c-8eec-06fe82dd4b17","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.379494Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:dd19f3f811ad4fef4bdc6c958d31ae3d16040e1c3e933faab3f1a699c9122f60","observation_id":"0f1e8a9f-4ec7-4a60-9e96-f23fc3968f19","resolution":{"observed_at":"2026-08-06T20:57:51.365506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.347821Z","title":"Emergence in non-neural models: grokking modular arithmetic via average gradient outer product","venue":null,"work_id":"a2885b51-fc6b-4dbb-a411-b2206d5303d1","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.383332Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:370e5e0a226d526cd7173c41a75539aec65253dc4bf8eec591e1fd8c6b651a2f","observation_id":"b811f6a1-233f-4b67-9aa6-3e190e35d328","resolution":{"observed_at":"2026-08-06T20:57:51.352256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.334516Z","title":"Hoffman, and David M","venue":null,"work_id":"a88e7619-d1f0-46b4-b5bb-46a3d9f7927c","year":2017},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.387426Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7e6c8b7b6b7896c37f277b5024cc7126622c3b0a4a0af6c5b91fa711d7deb1f0","observation_id":"2e476f30-2f64-4823-8ab4-256fca7fa2b9","resolution":{"observed_at":"2026-08-06T20:57:51.338471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"math-ph/0609050","last_updated":"2007-02-27T14:22:05Z","snapshot_observed_at":"2026-07-07T06:16:50.870997Z","submitted_at":"2006-09-18T11:18:42Z","title":"How to generate random matrices from the classical compact groups","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"math-ph/0609050","snapshot_observed_at":"2026-08-06T20:57:50.391686Z","title":"How to generate random matrices from the classical compact groups","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.391686Z"},"links":{"cited_paper":"/paper/math-ph/0609050","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e6410697ffcabf8733b9ff3bd45d4c79755549ce8a78443e0d2ea293aa033425","observation_id":"89ea9f58-4226-4de9-8aa3-64b1b117e973","resolution":{"observed_at":"2026-08-06T20:57:50.391686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.396097Z","title":"The quantization model of neural scaling.Advances in Neural Information Processing Systems, 36, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.396097Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d219a92620c578513c27e3fbb9e42012b229fb208135c7d9153c53e734b604d7","observation_id":"000462f9-ae81-4b07-b946-d478a9b70acd","resolution":{"observed_at":"2026-08-06T20:57:50.396097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.310350Z","title":"Rethinking the role of demonstrations: What makes in-context learning work?, 2022","venue":null,"work_id":"7c4291fe-f271-452c-a5d5-ff751d017625","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.399865Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:0a4edd48f51f3c907aa8ab650246ffd81ae35d8410aa88a07aa5576ec6acb7ff","observation_id":"b818bee2-639b-4d05-9836-953ecb2326b9","resolution":{"observed_at":"2026-08-06T20:57:51.314342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.296414Z","title":null,"venue":null,"work_id":"21ca1026-ab0d-475c-bd41-7fc81cbe9608","year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.404218Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9c15a77922b46f4771dff9a62885d05bc303614e09d9835dda3af6f87953b9a1","observation_id":"de8c6695-0166-46fa-bb2c-b0af4bfa9530","resolution":{"observed_at":"2026-08-06T20:57:51.301004Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.282471Z","title":"Grokking mod- ular arithmetic can be explained by margin maximization","venue":null,"work_id":"150f77f4-dfe0-40b8-84f3-efa89f6f72f9","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.408153Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9265bdd91319e9d8ed7ca390a63731941ed322a1ced0124ae7c7a8077abfbf43","observation_id":"12f57399-d789-4112-9fbb-12097e4db8c2","resolution":{"observed_at":"2026-08-06T20:57:51.286523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.269207Z","title":"Transformers can do bayesian inference, 2024","venue":null,"work_id":"dfcdae16-1906-4cec-84ca-9d14c2e56398","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.412570Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:21b38f089ed0f3cc9ede4b863888e79140e32c5ca26c74e31199a389352d5fbf","observation_id":"c4136913-72b4-4aa7-80dc-acff20b83f9c","resolution":{"observed_at":"2026-08-06T20:57:51.272974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.256445Z","title":null,"venue":null,"work_id":"1065df3b-100b-40f1-b7b3-af919221b570","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.416531Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:52f65416c17cf98223159b753644a5cbf039bf1636f9a18bf2267533610dc08d","observation_id":"4afa61df-3135-4f40-9fbe-3d2fc9b9d5c5","resolution":{"observed_at":"2026-08-06T20:57:51.260246Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.242918Z","title":"Progress measures for grokking via mechanistic interpretability","venue":null,"work_id":"c2d516b5-5a03-4c3f-9036-fc2ca2ac4533","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.420860Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:26022dc063a7dff7b5c960188f5cae7844efbae4694b891a754281330b486930","observation_id":"39032d8a-d2b7-44af-8ad3-b2e3b7d80178","resolution":{"observed_at":"2026-08-06T20:57:51.247422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.229725Z","title":"Differential learning kinetics govern the transition from memorization to generalization during in-context learning, 2024","venue":null,"work_id":"42207c71-492e-41e5-aa93-e82844e48d27","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.424811Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:465aea115cf133c101feb812af79b5004efcea994bae38aaecd62216515217a0","observation_id":"5cba795c-5d46-425e-8f06-e0ed8037d815","resolution":{"observed_at":"2026-08-06T20:57:51.233720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.428875Z","title":"Lee, and Alberto Bietti","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.428875Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:0a4ba9d2069d4eb30e7e0c7ba0271e0de21b42a6c2b1d1bb916233f84e06c55f","observation_id":"dfae44a1-5038-4732-a3fd-6f2bae8a87f4","resolution":{"observed_at":"2026-08-06T20:57:50.428875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.206879Z","title":null,"venue":null,"work_id":"74916ea7-7b3c-449b-9b30-f862410f9ef9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.433106Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:83a4b644298b4483adf0741d57843a7ece9ba80ea45fe32dfa2b6010e83a2ea4","observation_id":"fc02881c-073f-4447-a58a-c6529843e325","resolution":{"observed_at":"2026-08-06T20:57:51.210591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-06T20:57:50.436894Z","title":"In-context learning and induction heads.arXiv preprint arXiv:2209.11895, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.436894Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:1fbcdd9f4085d841c39ffc5f68d73dfb2991ebcab4e468b33235d288a230b9d8","observation_id":"723e21ed-5bbb-494a-ba82-8819d30f6d07","resolution":{"observed_at":"2026-08-06T20:57:50.436894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.194197Z","title":"What in-context learning \"learns\" in-context: Disentangling task recognition and task learning, 2023","venue":null,"work_id":"54a43f14-70c0-4ba3-aebf-0a4e62f7b8b7","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.440839Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ec3389958aeba25500147441218c904fb7d2e98f27f19a85ce315a3a8f68e127","observation_id":"804ec645-8b21-45a2-8494-2d2f930f556e","resolution":{"observed_at":"2026-08-06T20:57:51.198025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.182016Z","title":"In-context learning through the bayesian prism, 2024","venue":null,"work_id":"e5a4f42f-1cb0-4e8d-870e-17c2f76c8c00","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.444624Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:aa0ead3dee5ffa084a3f0c6e16b2ff1fbe71a9b0582b9d7cc396ae184bfd1972","observation_id":"a31572cc-2f52-44f7-99eb-3f535e10d3d3","resolution":{"observed_at":"2026-08-06T20:57:51.185916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.169128Z","title":"Competition dynamics shape algorithmic phases of in-context learning, 2025","venue":null,"work_id":"ab6369be-afa6-4150-b631-f8f7994c617c","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.448214Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:3b305db4f9265eff82bc6a37f089dde32deb35db514cf7a682e846076018d5a0","observation_id":"30793b95-aaa8-4e66-ab42-9d3ac3c85cb2","resolution":{"observed_at":"2026-08-06T20:57:51.173103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.155984Z","title":"Gradient starvation: A learning proclivity in neural networks.Advances in Neural Information Processing Systems, 34:1256–1272, 2021","venue":null,"work_id":"5d2366f3-8561-4dd6-95a6-873faebf148a","year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.451939Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:f2a7bb33270cc5cb3d09e235cfba6d57f01f0fe6a0033c1446444fd1f3868f14","observation_id":"56e50786-2ff2-4886-9bf7-18a2bb1b7030","resolution":{"observed_at":"2026-08-06T20:57:51.160478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.141974Z","title":"Multi-scale feature learning dynamics: Insights for double descent","venue":null,"work_id":"36231feb-343a-444f-9b9e-e4b95636ab9e","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.456276Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e84f77483207cedc801c061def8b3803fa2ccae3e7719fda55b8bc6fef63b84e","observation_id":"1bd7c59a-1a78-4b51-b1df-733fd66ea5f2","resolution":{"observed_at":"2026-08-06T20:57:51.146280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.02177","last_updated":"2022-01-06T18:43:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-06T18:43:37Z","title":"Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.02177","snapshot_observed_at":"2026-08-06T20:57:50.459920Z","title":"Grokking: Gen- eralization beyond overfitting on small algorithmic datasets.arXiv preprint arXiv:2201.02177, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.459920Z"},"links":{"cited_paper":"/paper/2201.02177","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:bbb75044d5cf6c27ce924ed811e2cf9af78fa287e49de7fca8948ba3bea1ac44","observation_id":"0cfea553-1a99-4ff6-9f62-f81d01c3d2a3","resolution":{"observed_at":"2026-08-06T20:57:50.459920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04697","last_updated":"2025-05-19T11:02:53Z","snapshot_observed_at":"2026-08-05T11:14:36.679405Z","submitted_at":"2025-01-08T18:58:48Z","title":"Grokking at the Edge of Numerical Stability","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04697","snapshot_observed_at":"2026-08-06T20:57:50.463997Z","title":"Grokking at the edge of numerical stability.arXiv preprint arXiv:2501.04697, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.463997Z"},"links":{"cited_paper":"/paper/2501.04697","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:04a9a9d8936350dae3eaf79cb234539c59eae0f3005a9102325fb4ad1cf6dfe8","observation_id":"1333739d-8086-4ea3-b2b4-c1b15760c450","resolution":{"observed_at":"2026-08-06T20:57:50.463997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.128136Z","title":"Transformers on markov data: Constant depth suffices","venue":null,"work_id":"368e05cc-c36a-4d5a-b493-36eb8411505c","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.469390Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:498f014125c1c650cfcb281fa2f7d8c467cc0ec6901a0f8f187095d4c198779e","observation_id":"0788a3e0-1ddb-4c08-8369-c2c6cdd85b21","resolution":{"observed_at":"2026-08-06T20:57:51.131965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.114868Z","title":"Pretraining task diversity and the emergence of non-Bayesian in-context learning for regression.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"41d3b2f6-4875-40fa-9761-7d5bfcd471b9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.473168Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:51861d75ad923a0ab08ea81849a5334200f18316384cafb66db9b234a72522d9","observation_id":"c138f444-e289-4a3a-955f-a983ca9ba18d","resolution":{"observed_at":"2026-08-06T20:57:51.119108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.101046Z","title":"The mechanistic basis of data dependence and abrupt learning in an in-context classification task, 2023","venue":null,"work_id":"dcf30d2f-b24e-44ef-87ac-c40921afb039","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.477710Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:f47b117f2eae1f571355b0ce936b3defcb069a7b42562e008914a0ae030c19dd","observation_id":"d8816f7d-d4f4-44ca-93a9-b11c684ea924","resolution":{"observed_at":"2026-08-06T20:57:51.105229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.088487Z","title":null,"venue":null,"work_id":"d4e546a9-17a5-4857-9487-d30838eb36fa","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.482042Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:6f6958708d058630360764a9adf125bd1bf64644853943c3d90593bcf46f166a","observation_id":"3b117b78-3ca9-46d5-be1f-4c625d258c31","resolution":{"observed_at":"2026-08-06T20:57:51.092216Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.074514Z","title":"Are emergent abilities of large language models a mirage?Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"37025723-47d9-4ad8-be3f-567a11aa71cd","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.485665Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7b1823cd1bf242a79c09d7f0ccf2182ddb5a6196253ea42a32860f0108ffb97d","observation_id":"d0be3b76-8c60-4dca-9dfc-0f69584c4c0b","resolution":{"observed_at":"2026-08-06T20:57:51.079229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.060407Z","title":"I preliminaries","venue":null,"work_id":"3e341b72-aeed-476f-85ac-60790694426b","year":1999},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.489660Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d674b108d9850494dc866772f667e9ca9371c7d2e733fcdb4af491b06931ad36","observation_id":"c0b2b5b6-35a6-4236-92a8-1a58e2b0ade3","resolution":{"observed_at":"2026-08-06T20:57:51.064250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.046586Z","title":"The pitfalls of simplicity bias in neural networks.Advances in Neural Information Processing Systems, 33:9573–9585, 2020","venue":null,"work_id":"947596f0-975b-4a97-81ba-cfa2618bb024","year":2020},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.493585Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:34ac384f6f4de6bec74f9a27e4fffd1331f52411ca9a6c91c8aa3a32bb7e0ee6","observation_id":"242a903a-fc53-44cc-a146-c3c17f33ccab","resolution":{"observed_at":"2026-08-06T20:57:51.051341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.033317Z","title":"Singh, Ted Moskovitz, Sara Dragutinovic, Felix Hill, Stephanie C","venue":null,"work_id":"cc08632f-65e2-4260-9d1c-2b8420eb97c0","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.497528Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fdd36157202ea0a8d6c440ae424265a31a26fbd9cbddebd2f5b22ba85da849b7","observation_id":"9c6ec1f4-3189-431b-867b-8c21c43043e3","resolution":{"observed_at":"2026-08-06T20:57:51.037428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.020116Z","title":"Singh, Ted Moskovitz, Felix Hill, Stephanie C","venue":null,"work_id":"300aa5e7-b2f8-4055-9e58-212e49098bd4","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.500971Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c15e22022215fc992e9784d17a8b5184cee7b1f8acee7f95d7582e1df2d062a6","observation_id":"2c1c3dee-e0cc-4819-bf45-53af02abf929","resolution":{"observed_at":"2026-08-06T20:57:51.024189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.006499Z","title":"The implicit bias of gradient descent on separable data.Journal of Machine Learning Research, 19:1–57, 2018","venue":null,"work_id":"0285c27e-a834-425d-b7be-2a90b70bcd7a","year":2018},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.504540Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fd18c27d646965bdc50a9b5c3abc8d5dde3fb7b53604896e5ba4a79e9d9a40cf","observation_id":"a103b708-0a07-4d13-bc18-0ea7a8e788ec","resolution":{"observed_at":"2026-08-06T20:57:51.010579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.508348Z","title":"Transformers learn in-context by gradient descent","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.508348Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fe8d3fc53f1774ed6fbc4ef9c94126339333fe9481757bb4b49d572e8dc77217","observation_id":"c891c0cf-7476-4b69-955f-2c8bae446104","resolution":{"observed_at":"2026-08-06T20:57:50.508348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.984343Z","title":"Label words are anchors: An information flow perspective for understanding in-context learning, 2023","venue":null,"work_id":"53975480-460a-4e01-a840-77afcbe2fd92","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.512389Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:76d1c90da6d52b2b5421d174710751a32b52fbd178e19cafb67f3233374f3dd1","observation_id":"dfe94556-f4bf-4227-9d1b-a21d3519d5a1","resolution":{"observed_at":"2026-08-06T20:57:50.988282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.971609Z","title":"Investigating the pre-training dynamics of in-context learning: Task recognition vs","venue":null,"work_id":"edfae7be-2d3b-4707-a56a-505a3c5a9993","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.516098Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:299913d7abedc95393d920153fbe684e58dbb87e44e83961d8d891c6cb75f5bb","observation_id":"7d418978-8f03-473f-8a2a-27b221e519d0","resolution":{"observed_at":"2026-08-06T20:57:50.975604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.958201Z","title":"Cambridge monographs on applied and computational mathematics ; 25","venue":null,"work_id":"1db6a00c-7334-46b2-922b-3fe85a9c827a","year":2009},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.519925Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:881e71692afaed7564d327af7fa856b4f691f5616cf74e43d9128636e60cf34c","observation_id":"31f1e21a-5422-4bd8-9538-0f6d41977210","resolution":{"observed_at":"2026-08-06T20:57:50.962394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.523610Z","title":"Emergent abilities of large language models.Transactions on Machine Learning Research, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.523610Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c225f81cc5a969978c6ec9916e2dc85bee83e3641d6566e25c0d754279229fbc","observation_id":"7f2b0d05-542c-44dd-9d49-b6bb9ca53394","resolution":{"observed_at":"2026-08-06T20:57:50.523610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.934459Z","title":"Symbol tuning improves in-context learning in language models","venue":null,"work_id":"dcb89470-3ab5-4954-8a70-dcfcc67fa2bd","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.527132Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e1130db1a9a1bb17f7b23bb1d8f7b295e7bc6bf85bf25c7d52d85acb28c96a84","observation_id":"89e7c1c8-f9d5-48f1-8ca4-3c063d7021c9","resolution":{"observed_at":"2026-08-06T20:57:50.938971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.920072Z","title":"Larger language models do in-context learning differently, 2023","venue":null,"work_id":"356f1526-1c4a-4fbe-ab02-cc878185a323","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.530600Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:28beb6b77a1461eeca42c770ea76f70efe5cfb98bbaaf5da98d0d79bbfdc5485","observation_id":"c235e701-5b2f-423d-9b9b-ab38e5b64c8a","resolution":{"observed_at":"2026-08-06T20:57:50.924799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.904822Z","title":"The learnability of in-context learning, 2023","venue":null,"work_id":"cabbd6c7-45e6-4877-82b8-2307d983d8ff","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.534207Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:cd0569afe6056a10e7c7fd2e744ab8a64eea0f8e240bb28525369b500506315f","observation_id":"4f03d646-60e7-4d00-bb02-6a3de0a0c218","resolution":{"observed_at":"2026-08-06T20:57:50.909234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.889339Z","title":"Bartlett","venue":null,"work_id":"13d59e70-5e81-41ae-946f-aa42071ea961","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.538652Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:99827444ca9f0bee43e9dea7ab65df032b5f1e24d6bc600a6796df6822f496d9","observation_id":"443613bf-e32b-4658-8855-fc8e5a97ae2c","resolution":{"observed_at":"2026-08-06T20:57:50.893332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02080","last_updated":"2022-07-21T07:44:13Z","snapshot_observed_at":"2026-07-30T03:45:35.558793Z","submitted_at":"2021-11-03T09:12:33Z","title":"An Explanation of In-context Learning as Implicit Bayesian Inference","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.02080","snapshot_observed_at":"2026-08-06T20:57:50.542232Z","title":"An explanation of in-context learning as implicit bayesian inference.arXiv preprint arXiv:2111.02080, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.542232Z"},"links":{"cited_paper":"/paper/2111.02080","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:705879d67abce35ef02ead9252c6048971fd25694aa1d2cf352cf6bcd5ce039b","observation_id":"9aa4c803-6d33-4dd4-81a1-bd1920753f32","resolution":{"observed_at":"2026-08-06T20:57:50.542232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.875523Z","title":"Which attention heads matter for in-context learning?, 2025","venue":null,"work_id":"9e7dd0ff-df84-4d02-ac18-c18c3ed991e4","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.545964Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:936c68e52b4c9f10219f3c20268d2df42eae7423c101569cfdaa34addfb934ac","observation_id":"8626323f-a16e-49be-9356-71764c22750d","resolution":{"observed_at":"2026-08-06T20:57:50.879642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1709.06493","last_updated":"2017-10-03T14:31:03Z","snapshot_observed_at":"2026-07-06T06:00:20.967053Z","submitted_at":"2017-09-19T15:55:16Z","title":"Learning to update Auto-associative Memory in Recurrent Neural Networks for Improving Sequence Memorization","version":3},"cited_work":{"arxiv_id":"1709.06493","doi":null,"metadata_source":"pith","pith_arxiv_id":"1709.06493","snapshot_observed_at":"2026-08-06T20:57:50.601484Z","title":"Learning to update Auto-associative Memory in Recurrent Neural Networks for Improving Sequence Memorization","venue":"cs.AI","work_id":"b15eb500-f2a4-4eb6-b082-a5fb3c8f3148","year":2017},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.549945Z"},"links":{"cited_paper":"/paper/1709.06493","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b66678494dd30d3a4e4ece1060cbfc444d936f2b4def99998296a811c8424212","observation_id":"ed80c7c0-bf72-4ce1-9cc5-baa668260520","resolution":{"observed_at":"2026-08-06T20:57:50.607902Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.860908Z","title":"Singh, Peter E","venue":null,"work_id":"94ecc74d-e7f9-48c6-94f7-c2faa5e15ec3","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.554227Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d16377649315457acc9cef471a88d8020088a99ea59fcaece179451b3e543cf1","observation_id":"c3428b09-90c6-4d74-81e6-aecdb2189132","resolution":{"observed_at":"2026-08-06T20:57:50.865767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.845945Z","title":"grokking","venue":null,"work_id":"698edad1-e1b9-4bb7-8ade-ea51d00fc6d9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.558118Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:56e31cfe0441cbbcf45b083c470dacb10b6d3c810b3b2735363e0255c049a785","observation_id":"e23d4f56-2c7b-436e-a12d-ccf86e2ee14d","resolution":{"observed_at":"2026-08-06T20:57:50.850211Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.832349Z","title":"tiny”, “small","venue":null,"work_id":"48043fc5-67f6-43a2-8b12-de967723f3fa","year":null},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.562642Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c0a93edcf6ec9978f3322fba6fa11b91272c50c0efcd9fc4e10caf3e24479753","observation_id":"7cf24358-5e8d-4b23-a0b3-ad11780b5920","resolution":{"observed_at":"2026-08-06T20:57:50.836438Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.723869Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.723869Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:56bf953810abcb1c28fbf43d76c4f3ee4aba203961c36049fb3abc1016f04396","observation_id":"1d1e9eaf-2462-474c-ba00-9998e195de43","resolution":{"observed_at":"2026-08-06T20:57:48.723869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T15:53:14.345071Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":2,"verified_fuzzy":47},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 0 inbound Pith citation observations for arXiv:2507.01414."}