{"as_of":"2026-08-09T20:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:abf9ddfccbfc6e170f4f2b57fbe639db92407aca37ceea46c0dfd769d379f666","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T13:44:00.474327Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":24,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2306.01116","last_updated":"2023-06-01T20:03:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-01T20:03:56Z","title":"The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data, and Web Data Only","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T20:43:45.770157Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2306.01116"},"observation_digest":"sha256:93ab9b0028435f74707778ff9365897ea9bc66e7035fe3c85fceacf9433b3c90","observation_id":"72709979-5cf3-409f-93c8-87bf672b1828","resolution":{"observed_at":"2026-05-13T20:43:45.805619Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2311.16867","last_updated":"2023-11-29T19:45:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-28T15:12:47Z","title":"The Falcon Series of Open Language Models","version":2},"reference_index":273,"source":"arxiv_source","source_observed_at":"2026-05-16T09:46:09.701440Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2311.16867"},"observation_digest":"sha256:fb2ad4607b441f7d8031ca86bec744102248e6dbcc5ec4db1fe9e9360a959bb7","observation_id":"11ce8e20-346e-4ee5-9948-e01920c03746","resolution":{"observed_at":"2026-05-16T09:46:10.073190Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T18:00:53.389420Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2404.06395"},"observation_digest":"sha256:dff1b4aa62331f338d69f315d7e6d9eff51b622b315144074a1f799af69f1cee","observation_id":"7ded08bb-9350-4ace-a73c-fa0057af11eb","resolution":{"observed_at":"2026-05-13T18:00:53.450722Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2501.05465","last_updated":"2026-05-14T16:52:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-03T19:53:57Z","title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T05:47:48.488826Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2501.05465"},"observation_digest":"sha256:48281d6bb95a22d5567e22d54a0bee6136d62bcb1ab977cda3ae1fe3a37d073d","observation_id":"7f2cbd0f-da97-422e-8f01-f1548ff192de","resolution":{"observed_at":"2026-05-23T05:52:37.423436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-08T13:44:00.474327Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.07832","last_updated":"2025-02-11T00:21:40Z","snapshot_observed_at":"2026-08-09T09:01:49.466932Z","submitted_at":"2025-02-11T00:21:40Z","title":"SHARP: Accelerating Language Model Inference by SHaring Adjacent layers with Recovery Parameters","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T13:44:00.474327Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2502.07832"},"observation_digest":"sha256:d81eefc8779564f20215f084ecf61ff33faa2d6a9469fd5e6aef9c3ee585f880","observation_id":"8d69a650-a8bd-4bfe-bada-f1598e01d261","resolution":{"observed_at":"2026-08-08T13:44:00.474327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T14:44:23.116345Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17716","last_updated":"2025-05-23T10:33:14Z","snapshot_observed_at":"2026-08-08T15:42:18.320838Z","submitted_at":"2025-05-23T10:33:14Z","title":"Get Experience from Practice: LLM Agents with Record & Replay","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:44:23.116345Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2505.17716"},"observation_digest":"sha256:87275c964d4d9f424f9431d5244e3385c5cd18644d6e08a560ff21301617ff91","observation_id":"0d10b42b-515c-40a5-b47a-0fafba6430f3","resolution":{"observed_at":"2026-08-07T14:44:23.116345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T12:45:26.558466Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23725","last_updated":"2026-06-02T01:19:21Z","snapshot_observed_at":"2026-08-07T12:36:41.591936Z","submitted_at":"2025-05-29T17:55:37Z","title":"MuLoCo: Muon is a practical inner optimizer for DiLoCo","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T12:45:26.558466Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2505.23725"},"observation_digest":"sha256:d12e49f9daeab3b4b29aa95f446413fa2674b4f658c5de5ad71314035386707b","observation_id":"df1ccc47-c817-46e5-b64c-4ed0e368cc4e","resolution":{"observed_at":"2026-08-07T12:45:26.558466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-07T05:50:18.794719Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06926","last_updated":"2025-06-07T21:29:25Z","snapshot_observed_at":"2026-08-09T16:57:33.722605Z","submitted_at":"2025-06-07T21:29:25Z","title":"Basis Transformers for Multi-Task Tabular Regression","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:50:18.794719Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2506.06926"},"observation_digest":"sha256:d4a89d36476d64ec58fe6c4d982a93ee5a2292a04202de768f02535295540050","observation_id":"7b19a67a-b969-4e9b-bd05-f3e374efaaa7","resolution":{"observed_at":"2026-08-07T05:50:18.794719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-06T11:44:04.928577Z","title":"Cerebras-gpt: Open compute-optimal language models trained on the cerebras wafer-scale cluster, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22448","last_updated":"2025-07-30T07:55:33Z","snapshot_observed_at":"2026-08-09T19:06:02.138099Z","submitted_at":"2025-07-30T07:55:33Z","title":"Falcon-H1: A Family of Hybrid-Head Language Models Redefining Efficiency and Performance","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T11:44:04.928577Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2507.22448"},"observation_digest":"sha256:9d45fac17de67e925642670ef33aa91e4641ba4f0cfed7563dde5969d727e4f4","observation_id":"4c52a78a-a6e0-4ad8-a59b-a3e36ad36ad0","resolution":{"observed_at":"2026-08-06T11:44:04.928577Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2509.18218","last_updated":"2026-04-06T17:04:13Z","snapshot_observed_at":"2026-07-06T22:30:31.933240Z","submitted_at":"2025-09-21T22:34:00Z","title":"Similarity Field Theory: A Mathematical Framework for Intelligence","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-18T14:36:49.543497Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2509.18218"},"observation_digest":"sha256:e56487e50254a0cca414f0ef81b9815a17a9ee1cce9a37c3ef000f5f396c1844","observation_id":"696b93b9-96f5-49dc-af04-c4a478f741f8","resolution":{"observed_at":"2026-05-18T14:41:30.649201Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2511.09447","last_updated":"2026-04-27T09:15:08Z","snapshot_observed_at":"2026-07-06T22:35:37.039024Z","submitted_at":"2025-11-12T16:03:52Z","title":"SpaDA: A Spatial Dataflow Architecture Programming Language","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T22:18:38.464461Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2511.09447"},"observation_digest":"sha256:dc6a04777f3ade63ef93e1dcbb28d075a8ceb0b390b9aacc65df23d106adaedb","observation_id":"f6d2d069-c6f6-451d-9f9d-23f49452f5d0","resolution":{"observed_at":"2026-05-17T22:20:23.248975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2603.00541","last_updated":"2026-05-11T13:53:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-28T08:38:50Z","title":"Spectral Condition for $\\mu$P under Width-Depth Scaling","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T18:03:31.202134Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2603.00541"},"observation_digest":"sha256:0559eed8e738d4c97b518f1f0c2f18898d8e5fe56c9a2668b1dd293fd0e1aa65","observation_id":"1c9812f9-a439-4926-ad6d-d58639597ea3","resolution":{"observed_at":"2026-05-15T18:06:25.514696Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.03199","last_updated":"2026-04-03T17:17:51Z","snapshot_observed_at":"2026-08-02T10:19:39.008617Z","submitted_at":"2026-04-03T17:17:51Z","title":"Learning the Signature of Memorization in Autoregressive Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T19:53:10.396785Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.03199"},"observation_digest":"sha256:fb276a0fecbce35d90c48b4810e101b699ce1e50b94b79af7cc4b8d939dcdb83","observation_id":"3030597b-90d3-49d4-b542-8bcf4d65b7e0","resolution":{"observed_at":"2026-05-13T19:53:11.472570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.04722","last_updated":"2026-04-06T14:45:49Z","snapshot_observed_at":"2026-07-06T22:53:37.357996Z","submitted_at":"2026-04-06T14:45:49Z","title":"Don't Waste Bits! Adaptive KV-Cache Quantization for Lightweight On-Device LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T19:31:17.420639Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.04722"},"observation_digest":"sha256:8be7d62ca00e12c86de1ae1ecca8c24c4f001435e98db0fd1fa5791dd40bab42","observation_id":"397fe58b-91bd-4d8c-832d-adabb3ac28a1","resolution":{"observed_at":"2026-05-10T22:55:48.072540Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2604.13275","last_updated":"2026-04-14T20:12:05Z","snapshot_observed_at":"2026-07-06T23:01:19.191464Z","submitted_at":"2026-04-14T20:12:05Z","title":"Better and Worse with Scale: How Contextual Entrainment Diverges with Model Size","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:18:33.458143Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2604.13275"},"observation_digest":"sha256:dcd96d433c4342b91a332ab58bcd386f406e47a70768aac1cc3770ad996c921a","observation_id":"a5243836-dd44-4be2-b867-29d28b90e009","resolution":{"observed_at":"2026-05-11T10:46:06.499168Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.15290","last_updated":"2026-05-14T18:03:16Z","snapshot_observed_at":"2026-07-06T23:26:37.362117Z","submitted_at":"2026-05-14T18:03:16Z","title":"GQA-{\\mu}P: The maximal parameterization update for grouped query attention","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T16:35:40.231293Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.15290"},"observation_digest":"sha256:1ac4c34f11fc1d366a4935b5a3c1b6ba0d0e3247f43add87e28d843820c49ae0","observation_id":"e53f190d-6c90-453b-8425-d49eb776e48f","resolution":{"observed_at":"2026-05-19T16:37:39.819716Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.15413","last_updated":"2026-05-14T20:57:15Z","snapshot_observed_at":"2026-08-08T21:55:40.844362Z","submitted_at":"2026-05-14T20:57:15Z","title":"Transformer Scalability Crisis: The First Comprehensive Empirical Analysis of Performance Walls in Modern Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-19T16:10:52.582492Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.15413"},"observation_digest":"sha256:d22c8908dc0248eda622906a7bca4dc6213968608acdc5bbbf1e385d4ded34a9","observation_id":"33fd6283-7502-4d89-822b-da9af1e96e67","resolution":{"observed_at":"2026-05-19T16:12:38.900193Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2605.29448","last_updated":"2026-07-18T01:59:39Z","snapshot_observed_at":"2026-08-06T23:52:37.250294Z","submitted_at":"2026-05-28T06:40:29Z","title":"How Much Is a Dataset Worth? Scaling Laws, the Vendi Score, and Matrix Spectral Functions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-29T08:33:05.952601Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.29448"},"observation_digest":"sha256:8e36b331b86f251d3d597071d72b478998e7d5004b586eddf16bf11ec8a5f044","observation_id":"e9e163fd-e946-4b77-a4fc-472196530f2d","resolution":{"observed_at":"2026-06-29T08:33:14.808607Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-02T12:58:43.750585Z","title":"et al.Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.29448","last_updated":"2026-07-18T01:59:39Z","snapshot_observed_at":"2026-08-06T23:52:37.250294Z","submitted_at":"2026-05-28T06:40:29Z","title":"How Much Is a Dataset Worth? Scaling Laws, the Vendi Score, and Matrix Spectral Functions","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T12:58:43.750585Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2605.29448"},"observation_digest":"sha256:2d88599a274c659567d958e847a96badccccb1ddcc2a8940020efc58f2461867","observation_id":"fd334a5c-971e-46d0-917f-5e35d581e4cb","resolution":{"observed_at":"2026-08-02T12:58:43.750585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.05029","last_updated":"2026-06-03T15:57:42Z","snapshot_observed_at":"2026-08-01T16:51:47.025025Z","submitted_at":"2026-06-03T15:57:42Z","title":"Validity Threats for Foundation Model Research","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T06:52:41.653304Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.05029"},"observation_digest":"sha256:c2f83e6e9522c2e3b309589223a05b71df136f98eb70c448c8a1fdca009d8f75","observation_id":"e4d79cee-fbab-475c-a54c-8e6d81140666","resolution":{"observed_at":"2026-07-02T07:36:44.918374Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.05610","last_updated":"2026-06-04T02:32:11Z","snapshot_observed_at":"2026-07-06T23:45:33.157370Z","submitted_at":"2026-06-04T02:32:11Z","title":"Predictable Scaling Laws of Optimal Hyperparameters for LLM Continued Pre-training","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-28T01:53:04.715108Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.05610"},"observation_digest":"sha256:4c7642fbb8367bab313bf075f020f4439c5c99ec11ab438a3447cbee5ceb9e1c","observation_id":"83636f29-a75c-4049-8d5e-065e442f67a7","resolution":{"observed_at":"2026-07-02T12:46:56.727189Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.20299","last_updated":"2026-07-01T14:03:37Z","snapshot_observed_at":"2026-08-09T16:51:41.612467Z","submitted_at":"2026-06-18T14:35:53Z","title":"Statistical Properties of Training & Generalization","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-06-26T15:35:51.654392Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.20299"},"observation_digest":"sha256:32038c261a762bca8dbeab6d7492971cbd5af33950db2591dddd615b0630b6b9","observation_id":"345089cb-143d-4052-8b8a-99f84689ca1e","resolution":{"observed_at":"2026-06-26T15:39:33.213970Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.20299","last_updated":"2026-07-01T14:03:37Z","snapshot_observed_at":"2026-08-09T16:51:41.612467Z","submitted_at":"2026-06-18T14:35:53Z","title":"Statistical Properties of Training & Generalization","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-07-02T21:51:13.457071Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.20299"},"observation_digest":"sha256:7c7f81c2a16d2427da3546f5ddcbb446d2e7034f90db54373f0fe78c3a840fbd","observation_id":"4b39796f-3374-4299-852e-1e98a5359572","resolution":{"observed_at":"2026-07-02T21:57:25.386062Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-02T23:29:21.699637Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-26T09:19:50.623741Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:8786389f2d769a480f42b0f5485d47253d2e1d7c44ba19e6119b912195f4b7ff","observation_id":"be861fad-ea9a-493a-834d-189e4735834a","resolution":{"observed_at":"2026-07-04T09:59:45.114260Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.22873","last_updated":"2026-06-25T18:44:01Z","snapshot_observed_at":"2026-08-02T23:29:21.699637Z","submitted_at":"2026-06-22T05:37:43Z","title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-29T01:18:19.195007Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.22873"},"observation_digest":"sha256:a1fe3ccdfd36178c1ef90772eabc74a2f00c9182f19a4be8af9cdda019b2c547","observation_id":"7eccc6d1-5602-4853-ae63-1f8ab990bf7f","resolution":{"observed_at":"2026-07-01T18:55:59.707947Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster","version":1},"cited_work":{"arxiv_id":"2304.03208","doi":"10.48550/arxiv.2304.03208","metadata_source":"arxiv_reference","pith_arxiv_id":"2304.03208","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"S., Chen, Z., Khachane, H., Marshall, W., Pathria, R., Tom, M., and Hestness, J","venue":"arXiv (Cornell University)","work_id":"146665a8-e124-465b-aa7f-84693124b019","year":2023},"citing_paper":{"arxiv_id":"2606.24077","last_updated":"2026-06-23T02:42:33Z","snapshot_observed_at":"2026-08-03T17:16:10.837184Z","submitted_at":"2026-06-23T02:42:33Z","title":"Sentence-Level Contextual Entrainment in Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T00:52:53.959215Z"},"links":{"cited_paper":"/paper/2304.03208","citing_paper":"/paper/2606.24077"},"observation_digest":"sha256:f7673013b35728fbd38fdd7e47814ba193746580eb9b227d555f2177e4d363e3","observation_id":"970f52ba-dd88-4742-ad29-5b29c5482ca2","resolution":{"observed_at":"2026-07-04T16:09:57.234685Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2304.03208/citation-record","integrity":"/paper/2304.03208/integrity","json":"/paper/2304.03208/citation-record.json","paper":"/paper/2304.03208"},"outbound":[],"paper":{"arxiv_id":"2304.03208","last_updated":"2023-04-06T16:43:16Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T15:13:02.077291Z","submitted_at":"2023-04-06T16:43:16Z","title":"Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cerebras Wafer-Scale Cluster"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2304.03208."}