{"as_of":"2026-08-09T13:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7da6b73bd75358bf9f9a3a67d9c4b8b7917252e9513e043c4d2db70abb4c1e77","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":27,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":27,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":27,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T20:25:25.286028Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-08T20:25:25.286028Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.286028Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:b9d824ab22d356a1a08821f504936ec579df6cf7c13ce4511b443fae356e0de0","observation_id":"8807e8bb-2dfb-425a-890a-c20c02444a30","resolution":{"observed_at":"2026-08-08T20:25:25.286028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-08T15:04:29.637076Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06604","last_updated":"2025-05-16T03:43:21Z","snapshot_observed_at":"2026-08-08T23:02:01.951277Z","submitted_at":"2025-02-10T16:01:55Z","title":"Do we really have to filter out random noise in pre-training data for language models?","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-08T15:04:29.637076Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2502.06604"},"observation_digest":"sha256:6464de2069474335af57876f76c3fb34b8b46050acde5abf316e806ba5bbfc2b","observation_id":"53abca1b-6c74-4c3c-ba55-f33eee8ce64b","resolution":{"observed_at":"2026-08-08T15:04:29.637076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T15:42:59.545380Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14070","last_updated":"2025-05-31T03:47:36Z","snapshot_observed_at":"2026-08-09T03:52:59.052704Z","submitted_at":"2025-05-20T08:21:37Z","title":"Enhancing LLMs via High-Knowledge Data Selection","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T15:42:59.545380Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2505.14070"},"observation_digest":"sha256:7b2f5e952ce6a01d68313bf574f756c84c6f5ebcd899ac8549ab7a5a8049b370","observation_id":"dae8452d-0891-4b6e-8f99-e09d2dee3682","resolution":{"observed_at":"2026-08-07T15:42:59.545380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T14:31:43.086949Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18690","last_updated":"2025-05-24T13:32:03Z","snapshot_observed_at":"2026-08-09T04:26:26.329174Z","submitted_at":"2025-05-24T13:32:03Z","title":"Benchmarking and Rethinking Knowledge Editing for Large Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T14:31:43.086949Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2505.18690"},"observation_digest":"sha256:730eab1265e9ee9f987c75bcf69788e5d15c68ffbb35039278900f4f0a754b71","observation_id":"fac541c2-6b00-412c-af75-dd51c7efab8e","resolution":{"observed_at":"2026-08-07T14:31:43.086949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T13:47:24.755626Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws.arXiv preprint arXiv:2404.05405, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20993","last_updated":"2025-05-27T10:26:47Z","snapshot_observed_at":"2026-08-09T00:22:01.118886Z","submitted_at":"2025-05-27T10:26:47Z","title":"Who Reasons in the Large Language Models?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:47:24.755626Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2505.20993"},"observation_digest":"sha256:5d61b151db662dc2fd580df40c0fa9f28a80c323574608194e23fe9bf749e4a4","observation_id":"6f112d02-2bfd-4ede-9044-a566f22c6636","resolution":{"observed_at":"2026-08-07T13:47:24.755626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T12:35:35.752224Z","title":"and Li, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24832","last_updated":"2025-06-18T15:27:03Z","snapshot_observed_at":"2026-08-08T09:16:38.722740Z","submitted_at":"2025-05-30T17:34:03Z","title":"How much do language models memorize?","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:35.752224Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2505.24832"},"observation_digest":"sha256:ca7fff37d7644f33db087eb0cc04e46981258859360407219482c0d3f5d96831","observation_id":"230120bb-4c84-4e7c-ad86-ed7509406b8b","resolution":{"observed_at":"2026-08-07T12:35:35.752224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T11:57:22.571092Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01115","last_updated":"2025-09-03T19:18:47Z","snapshot_observed_at":"2026-08-08T11:54:01.449606Z","submitted_at":"2025-06-01T18:42:39Z","title":"Is Random Attention Sufficient for Sequence Modeling? Disentangling Trainable Components in the Transformer","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:57:22.571092Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2506.01115"},"observation_digest":"sha256:8eadeb10d2e8df0a744daca4ab73d709bfed12192f4bc9079022b136b5eed7a8","observation_id":"8363ba87-329c-42c9-9ecc-0a23659bea68","resolution":{"observed_at":"2026-08-07T11:57:22.571092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-07T00:55:16.928046Z","title":"Allen-Zhu, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12312","last_updated":"2025-06-14T02:22:28Z","snapshot_observed_at":"2026-08-09T04:53:53.558105Z","submitted_at":"2025-06-14T02:22:28Z","title":"Perspective on Utilizing Foundation Models for Laboratory Automation in Materials Research","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T00:55:16.928046Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2506.12312"},"observation_digest":"sha256:03cc1a098cf3530cd30266cc0cf105af79942a70de5214acd24b4d1d90cc6792","observation_id":"5b8cac3c-1757-41fe-92ee-44ad9b2ec53a","resolution":{"observed_at":"2026-08-07T00:55:16.928046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-06T20:20:40.991429Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03300","last_updated":"2025-07-04T05:10:20Z","snapshot_observed_at":"2026-08-08T21:21:24.114593Z","submitted_at":"2025-07-04T05:10:20Z","title":"LRM-1B: Towards Large Routing Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:20:40.991429Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2507.03300"},"observation_digest":"sha256:af6f04680fe7ec6ffc5e9947757507506b284f53bd672b9714501c024e58cd0d","observation_id":"12765b3c-a70b-4027-b554-92f4ff607306","resolution":{"observed_at":"2026-08-06T20:20:40.991429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2511.21613","last_updated":"2026-04-19T02:59:16Z","snapshot_observed_at":"2026-08-06T11:38:39.962564Z","submitted_at":"2025-11-26T17:36:31Z","title":"Beyond URLs: Metadata Diversity and Position for Efficient LLM Pretraining","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T04:46:33.641714Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2511.21613"},"observation_digest":"sha256:66278a1a918cd3af9822ba9e54ab22324d8e74eb7883d975aa694f813263fccc","observation_id":"a2d54293-664b-4967-962c-c01c8d13cfd6","resolution":{"observed_at":"2026-05-17T04:49:03.055714Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-03T15:22:46.790701Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.17351","last_updated":"2026-07-28T16:53:50Z","snapshot_observed_at":"2026-08-03T18:14:53.588030Z","submitted_at":"2025-12-19T08:47:28Z","title":"Physics of Language Models: Part 4.1, Architecture Design and the Magic of Canon Layers","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T15:22:46.790701Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2512.17351"},"observation_digest":"sha256:f9a0253f09df6b63fc83aa8d8958c0415b9e8ce6d2347bcd7bd455643abd8001","observation_id":"2f4a7128-fe57-4182-bffd-7867cd67e99d","resolution":{"observed_at":"2026-08-03T15:22:46.790701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-03T14:02:56.631725Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws.arXiv preprint arXiv:2404.05405,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.22088","last_updated":"2026-06-10T15:31:41Z","snapshot_observed_at":"2026-08-09T06:43:58.919201Z","submitted_at":"2025-12-26T17:20:09Z","title":"Unifying Learning Dynamics and Generalization in Transformers Scaling Law","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T14:02:56.631725Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2512.22088"},"observation_digest":"sha256:2ffa4997f561486e5663047283ddcdac0a6031cb895384f7592a0570a1157954","observation_id":"80db9ad1-86fd-48bb-916b-0bdcf700aa62","resolution":{"observed_at":"2026-08-03T14:02:56.631725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2603.26554","last_updated":"2026-04-28T07:36:37Z","snapshot_observed_at":"2026-07-06T22:50:46.243903Z","submitted_at":"2026-03-27T16:13:18Z","title":"Sharp Capacity Scaling of Spectral Optimizers in Learning Associative Memory","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-14T23:37:33.106390Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2603.26554"},"observation_digest":"sha256:20b26170799cf310a915d8cedf3b0fcfafc59cf8bc6b028a037cfca8b41638d2","observation_id":"bb00fe68-5dc0-4c74-ba4e-e5b3d945a46c","resolution":{"observed_at":"2026-05-14T23:38:16.515544Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2604.08519","last_updated":"2026-04-09T17:55:50Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-04-09T17:55:50Z","title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-10T17:42:31.465077Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2604.08519"},"observation_digest":"sha256:a4645a684c95237d506ac753910c19380bbf1556dede2682b92682d02bbdac8d","observation_id":"3ca16e85-c092-4cea-870a-802d59fa8e1a","resolution":{"observed_at":"2026-05-11T06:15:59.180654Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.05189","last_updated":"2026-05-06T17:53:20Z","snapshot_observed_at":"2026-07-06T23:17:52.987860Z","submitted_at":"2026-05-06T17:53:20Z","title":"Sharp Capacity Thresholds in Linear Associative Memory: From Winner-Take-All to Listwise Retrieval","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-08T16:23:37.393165Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.05189"},"observation_digest":"sha256:693faca00f350123bfcbdfc718ad02d0d048c778f768e7e66ef0ef161fe51a35","observation_id":"c8b17a7a-e22b-4e42-bde3-aab89807ec82","resolution":{"observed_at":"2026-05-11T18:16:09.968352Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.09471","last_updated":"2026-05-10T10:56:59Z","snapshot_observed_at":"2026-08-05T20:10:28.703944Z","submitted_at":"2026-05-10T10:56:59Z","title":"The Statistical Cost of Adaptation in Multi-Source Transfer Learning","version":1},"reference_index":149,"source":"arxiv_source","source_observed_at":"2026-05-12T04:19:05.837824Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.09471"},"observation_digest":"sha256:bc52884b005f8239e4a264207631cd4948e0ba35384c9d9fb41b5f3c90786989","observation_id":"be8cea79-b02a-41a6-9ef6-3da7a203bc97","resolution":{"observed_at":"2026-05-12T04:21:20.732302Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.09724","last_updated":"2026-05-10T19:47:39Z","snapshot_observed_at":"2026-07-06T23:21:44.900680Z","submitted_at":"2026-05-10T19:47:39Z","title":"Model Capacity Determines Grokking through Competing Memorisation and Generalisation Speeds","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-12T03:55:28.036044Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.09724"},"observation_digest":"sha256:33aebe9ce8c6ee7b058efd64b3ce521357b027e344e047d6572e99c3bdae0146","observation_id":"6c2c10a4-5353-46a6-9997-5f52857e7eb0","resolution":{"observed_at":"2026-05-12T03:56:21.694097Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.10129","last_updated":"2026-05-11T07:40:39Z","snapshot_observed_at":"2026-07-06T23:22:08.560919Z","submitted_at":"2026-05-11T07:40:39Z","title":"Synthetic Pre-Pre-Training Improves Language Model Robustness to Noisy Pre-Training Data","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-12T03:38:39.738401Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.10129"},"observation_digest":"sha256:db221a1b5f5b5f0f9dae15f5660932a9c5bb89864c1f8be173f04b8810711f12","observation_id":"ab0cf40e-d3e9-400d-8816-7ab6d0e3cd69","resolution":{"observed_at":"2026-05-12T07:11:25.645042Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.12426","last_updated":"2026-05-12T17:22:22Z","snapshot_observed_at":"2026-07-30T10:50:56.815226Z","submitted_at":"2026-05-12T17:22:22Z","title":"Geometric Factual Recall in Transformers","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-13T04:50:17.479689Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.12426"},"observation_digest":"sha256:5542aa9869482a229353212eecbb291a5506bf5032f8bf6fa3216b1261ec301e","observation_id":"a3c588df-af3e-468f-b981-11cb2ccf4843","resolution":{"observed_at":"2026-05-13T04:52:16.765589Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.18732","last_updated":"2026-05-18T17:53:44Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:53:44Z","title":"Predictable Confabulations: Factual Recall by LLMs Scales with Model Size and Topic Frequency","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T10:59:32.371351Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.18732"},"observation_digest":"sha256:12af51aebb8b29ee01eff68de8235115b08ca3fae9d4eb502c251383a62e403b","observation_id":"b2ddf72e-668a-46af-9637-d1fe8700ced0","resolution":{"observed_at":"2026-05-20T11:03:13.729220Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2605.26097","last_updated":"2026-05-25T17:54:34Z","snapshot_observed_at":"2026-07-06T23:36:01.564747Z","submitted_at":"2026-05-25T17:54:34Z","title":"Forgetting in Language Models: Capacity, Optimization, and Self-Generated Replay","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T23:15:45.174086Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2605.26097"},"observation_digest":"sha256:2dd669092d24d13a13676647a75409bde0a278f43e7ea66b95a3ddc08b0d4296","observation_id":"361ac1e7-15c4-419e-9356-0939618aa96a","resolution":{"observed_at":"2026-06-29T23:24:02.010186Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2606.00570","last_updated":"2026-05-30T06:44:40Z","snapshot_observed_at":"2026-08-03T04:12:33.083328Z","submitted_at":"2026-05-30T06:44:40Z","title":"Revisiting Parameter-Based Knowledge Editing in Large Language Models: Theoretical Limits and Empirical Evidence","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-28T19:03:00.055800Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2606.00570"},"observation_digest":"sha256:8c993eab6a171ab6cb52d09ccefbaafb248124e264569f7142a07dc56c3b2e09","observation_id":"30009f3b-5a48-4c83-89a9-bfd29aeec2b5","resolution":{"observed_at":"2026-06-28T19:32:35.384173Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2606.03142","last_updated":"2026-06-02T04:36:32Z","snapshot_observed_at":"2026-08-08T17:22:57.853112Z","submitted_at":"2026-06-02T04:36:32Z","title":"Disentangling Visual and Factual Correctness in LVLMs' Visualization Literacy","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-28T11:19:51.893740Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2606.03142"},"observation_digest":"sha256:a2703a406ed61715cbe420bf83676d846a98b449427d2688cb4bbe52498ab10b","observation_id":"34fcdca2-4fc8-4dec-8db4-b42631fa4d57","resolution":{"observed_at":"2026-07-02T02:06:26.194790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2606.04662","last_updated":"2026-06-03T09:40:30Z","snapshot_observed_at":"2026-07-06T23:44:42.700120Z","submitted_at":"2026-06-03T09:40:30Z","title":"Why Muon Outperforms Adam: A Curvature Perspective","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-06-28T07:04:21.012269Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2606.04662"},"observation_digest":"sha256:f0eae5e0480aa6c38925107351e4b656ca6b1ccb8873e5f5c54048896b326c4d","observation_id":"d793f814-5ab8-4bff-9e5c-0c5c149ff787","resolution":{"observed_at":"2026-07-02T07:06:44.968032Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":"2404.05405","doi":"10.48550/arxiv.2404.05405","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":"arXiv (Cornell University)","work_id":"316d7b8f-344c-4c99-a387-d012a17b5c50","year":2025},"citing_paper":{"arxiv_id":"2606.19172","last_updated":"2026-06-17T15:15:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-17T15:15:19Z","title":"User as Engram: Internalizing Per-User Memory as Local Parametric Edits","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-06-26T20:37:01.382431Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2606.19172"},"observation_digest":"sha256:2eaa7004514e5db5933d10af0e7dd1464923686ca34c92c8b40d5a887c950cac","observation_id":"208f2e23-d01e-413c-8bc4-d1050aa7904b","resolution":{"observed_at":"2026-07-04T01:09:19.521587Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-02T05:21:29.896353Z","title":"and Li, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13432","last_updated":"2026-07-15T04:31:20Z","snapshot_observed_at":"2026-08-09T11:02:54.430128Z","submitted_at":"2026-07-15T04:31:20Z","title":"Local Redundancy: An Information-Theoretic Measure of Plasticity from Synthetic Memorization","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-02T05:21:29.896353Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2607.13432"},"observation_digest":"sha256:37936bba68d59b24c1a025d70062fc30f2fa657c4d9942eb736308c8d9ec77c9","observation_id":"5691762a-6d23-4559-9c4a-454c35006e6d","resolution":{"observed_at":"2026-08-02T05:21:29.896353Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-01T06:29:14.690542Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.21861","last_updated":"2026-07-23T23:14:59Z","snapshot_observed_at":"2026-08-07T01:31:36.362218Z","submitted_at":"2026-07-23T23:14:59Z","title":"Data Quality over Capacity: Internalizing Documents into LoRA Adapters for Closed-Book QA","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-01T06:29:14.690542Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2607.21861"},"observation_digest":"sha256:d6ce05fb346467ddfc44a81ba151c1784463bf73c07e20e8b05e36f200cf3ab9","observation_id":"77a5b3a1-41fd-43c3-a040-0b17199b8e21","resolution":{"observed_at":"2026-08-01T06:29:14.690542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2404.05405/citation-record","integrity":"/paper/2404.05405/integrity","json":"/paper/2404.05405/citation-record.json","paper":"/paper/2404.05405"},"outbound":[],"paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 27 inbound Pith citation observations for arXiv:2404.05405."}