{"as_of":"2026-08-10T12:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:118375cbe2c72962e017bff8b245ec28439ff4758626363c140227a7bd3e0019","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T17:09:26.809525Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":168,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2204.02311","last_updated":"2022-10-05T06:02:24Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-05T16:11:45Z","title":"PaLM: Scaling Language Modeling with Pathways","version":5},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-10T23:45:06.755839Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2204.02311"},"observation_digest":"sha256:89356fa957a83ae377fcff14359494902b7386b2429ae4c2a324259f308cb41f","observation_id":"f6ac1ab7-2b5c-4907-ab43-e42737955352","resolution":{"observed_at":"2026-05-10T23:45:07.185970Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2206.07682","last_updated":"2022-10-26T05:06:24Z","snapshot_observed_at":"2026-08-02T15:56:35.249569Z","submitted_at":"2022-06-15T17:32:01Z","title":"Emergent Abilities of Large Language Models","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-11T07:38:37.734402Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2206.07682"},"observation_digest":"sha256:900d74d4f2ccaf3f8a4a4cc59d031afbb8b41014fbf7f5fe557389f9c6ba365a","observation_id":"40bc1e35-4afd-4601-a92a-77bdfc24d024","resolution":{"observed_at":"2026-05-11T07:38:38.291396Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2206.10789","last_updated":"2022-06-22T01:11:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-06-22T01:11:29Z","title":"Scaling Autoregressive Models for Content-Rich Text-to-Image Generation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T04:49:30.873360Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2206.10789"},"observation_digest":"sha256:844b106d4d4239e82c0e8e5bda194e3b773bcd9e52443a6ab49f1c4cfa3d6ecd","observation_id":"545b5eed-f442-4098-ae5f-386a3b60a53f","resolution":{"observed_at":"2026-05-12T04:49:31.100632Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2207.14255","last_updated":"2022-07-28T17:40:47Z","snapshot_observed_at":"2026-08-07T11:38:07.397956Z","submitted_at":"2022-07-28T17:40:47Z","title":"Efficient Training of Language Models to Fill in the Middle","version":1},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-05-18T00:40:41.647820Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2207.14255"},"observation_digest":"sha256:495f0a4debee96f5f5e65f6d364e27155896d5b8b64e04e658b4e10e542bd02a","observation_id":"3d053dd6-2f60-46aa-928b-b0d7114fc72d","resolution":{"observed_at":"2026-05-18T00:40:41.895334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2305.10403","last_updated":"2023-09-13T20:35:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-17T17:46:53Z","title":"PaLM 2 Technical Report","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-12T11:59:25.813128Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2305.10403"},"observation_digest":"sha256:37ed13926b68211a91afff4adf47d93a4b5fde1fa606dd126929ed7798584cdf","observation_id":"470adb7a-1458-442b-8356-efc98aa84487","resolution":{"observed_at":"2026-05-12T11:59:27.202634Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-04T19:27:31.715261Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"reference_index":169,"source":"arxiv_source","source_observed_at":"2026-05-13T01:07:22.166595Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2309.14509"},"observation_digest":"sha256:0a7dc1f6794d0ff0ead0617635da4d3a436dadee4577d4ca0a0916351ae2b517","observation_id":"9fcfa9f6-ed31-42d1-86f3-3925a5abbb97","resolution":{"observed_at":"2026-05-13T01:07:22.297557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-09T17:09:26.809525Z","title":"Preprint, arXiv:2112.06905","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.00965","last_updated":"2025-05-25T17:39:32Z","snapshot_observed_at":"2026-08-09T17:29:31.414202Z","submitted_at":"2025-02-03T00:04:50Z","title":"CLIP-UP: A Simple and Efficient Mixture-of-Experts CLIP Training Recipe with Sparse Upcycling","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-09T17:09:26.809525Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2502.00965"},"observation_digest":"sha256:51422e15842b6a16886c339d8949e4c025ee25b6c3b2498699da47543e8ab064","observation_id":"750675db-14cd-4cb7-83bb-84120156c793","resolution":{"observed_at":"2026-08-09T17:09:26.809525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-09T13:40:56.440954Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.02040","last_updated":"2025-02-04T06:13:52Z","snapshot_observed_at":"2026-08-09T14:19:20.730533Z","submitted_at":"2025-02-04T06:13:52Z","title":"M2R2: Mixture of Multi-Rate Residuals for Efficient Transformer Inference","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T13:40:56.440954Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2502.02040"},"observation_digest":"sha256:374ea653bb85c7c2533983ea81b979e3b752409bc45200ab6177034aa63822cd","observation_id":"f6e487e9-a792-4216-8855-6e30d3dee94a","resolution":{"observed_at":"2026-08-09T13:40:56.440954Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-07T13:33:49.759942Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.21598","last_updated":"2025-05-27T16:56:54Z","snapshot_observed_at":"2026-08-09T06:28:29.015364Z","submitted_at":"2025-05-27T16:56:54Z","title":"Rethinking Data Mixture for Large Language Models: A Comprehensive Survey and New Perspectives","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T13:33:49.759942Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2505.21598"},"observation_digest":"sha256:d44c4c498ed383fa0f821064eb04805fdf1b881187f2dab200afc5441f10f7bb","observation_id":"963250a7-090d-4ab7-9b92-d715f9f2a4b9","resolution":{"observed_at":"2026-08-07T13:33:49.759942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-07T11:18:58.233961Z","title":"arXiv: 2112.06905 [cs.CL]","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02890","last_updated":"2025-06-03T13:55:48Z","snapshot_observed_at":"2026-08-09T14:19:36.325576Z","submitted_at":"2025-06-03T13:55:48Z","title":"Scaling Fine-Grained MoE Beyond 50B Parameters: Empirical Evaluation and Practical Insights","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:18:58.233961Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2506.02890"},"observation_digest":"sha256:9788cebf702b5cf125421e74fa209e977cbc048f9412e0ab21062c4eddec5215","observation_id":"ac690ccf-7837-4ab3-86b4-91ad969c3ef0","resolution":{"observed_at":"2026-08-07T11:18:58.233961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-07T10:30:34.175386Z","title":"Dai, Simon Tong, Dmitry Lepikhin, Yuanzhong Xu, Dehao Chen, Yonghui Wu, and Jeff Dean","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05333","last_updated":"2025-06-20T01:25:25Z","snapshot_observed_at":"2026-08-07T10:18:59.977399Z","submitted_at":"2025-06-05T17:59:24Z","title":"Kinetics: Rethinking Test-Time Scaling Laws","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:30:34.175386Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2506.05333"},"observation_digest":"sha256:e25af8b44e4de4d488abc88b21ae8e84c3e88cde5e4e265fa6c84c541a3f77d7","observation_id":"d1bb6067-0114-418b-88a9-1221a291043c","resolution":{"observed_at":"2026-08-07T10:30:34.175386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-19T05:48:02.828938Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2507.06261"},"observation_digest":"sha256:042babf00e239a2bb2df18afe0a49e9596f6d6ba1565074ceda891f47e4ca3a3","observation_id":"35e6ba7b-1919-4fa1-9b66-094c5418f6e7","resolution":{"observed_at":"2026-05-19T05:52:07.921973Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-06T18:24:51.180698Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08567","last_updated":"2025-08-07T11:18:14Z","snapshot_observed_at":"2026-08-09T17:00:22.878553Z","submitted_at":"2025-07-11T13:11:11Z","title":"AbbIE: Autoregressive Block-Based Iterative Encoder for Efficient Sequence Modeling","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:24:51.180698Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2507.08567"},"observation_digest":"sha256:bbb5b06c60a03801dc9f35bdfcc1859eb5653dd41401d55ced646c900d292626","observation_id":"33aae63d-db23-4645-aa10-691cc4d1adb6","resolution":{"observed_at":"2026-08-06T18:24:51.180698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-06T16:26:58.562199Z","title":"Xianzhi Du, Tom Gunter, Xiang Kong, Mark Lee, Zirui Wang, Aonan Zhang, Nan Du, and Ruoming Pang","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.13575","last_updated":"2025-08-27T16:34:47Z","snapshot_observed_at":"2026-08-09T12:02:36.199859Z","submitted_at":"2025-07-17T23:37:19Z","title":"Apple Intelligence Foundation Language Models: Tech Report 2025","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-06T16:26:58.562199Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2507.13575"},"observation_digest":"sha256:5f2f76b2e06321450f88949da2f0273b4793a15a079c3242388cc34cf9e0cc52","observation_id":"3cef134b-8794-431d-9533-d59c13455cff","resolution":{"observed_at":"2026-08-06T16:26:58.562199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-06T13:56:13.238051Z","title":"M., Tong, S., Lepikhin, D., Xu, Y.,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.20018","last_updated":"2025-07-29T05:18:46Z","snapshot_observed_at":"2026-08-10T00:19:13.677143Z","submitted_at":"2025-07-26T17:21:11Z","title":"The Carbon Cost of Conversation, Sustainability in the Age of Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T13:56:13.238051Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2507.20018"},"observation_digest":"sha256:440c51269cabad8f0ad6b05157f3ee91669b5b969f0b70015fb45091e16aadcf","observation_id":"0f324ac6-b139-4f9d-ad51-d0c8b51e4080","resolution":{"observed_at":"2026-08-06T13:56:13.238051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-04T22:55:28.421116Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.08852","last_updated":"2025-09-08T17:52:08Z","snapshot_observed_at":"2026-08-06T09:18:05.598392Z","submitted_at":"2025-09-08T17:52:08Z","title":"Safe and Certifiable AI Systems: Concepts, Challenges, and Lessons Learned","version":1},"reference_index":2006,"source":"pdf_text","source_observed_at":"2026-08-04T22:55:28.421116Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2509.08852"},"observation_digest":"sha256:b3a170201075ede1056d150d2d30da3dadff7a47f3f0c14a0dea4c73916eebd9","observation_id":"dff329e2-b034-45f8-86b2-a43ad82782b3","resolution":{"observed_at":"2026-08-04T22:55:28.421116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-04T14:50:26.916040Z","title":"GLaM: Efficient scaling of language models with mixture-of-experts.arXiv preprint arXiv:2112.06905,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.02345","last_updated":"2026-07-21T06:35:53Z","snapshot_observed_at":"2026-08-04T14:50:24.826478Z","submitted_at":"2025-09-27T10:45:58Z","title":"Breaking the MoE LLM Trilemma: Dynamic Expert Clustering with Structured Compression","version":4},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-04T14:50:26.916040Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2510.02345"},"observation_digest":"sha256:fe851cfaa3eada236e62f844ec873d768f19fefe97cfb00f42aad13833caca88","observation_id":"8cf3ad8a-9b48-4d79-aaee-355c18ad0f5c","resolution":{"observed_at":"2026-08-04T14:50:26.916040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-04T08:12:24.535811Z","title":"GLaM: Efficient scaling of language models with mixture-of-experts,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2510.22052","last_updated":"2026-07-09T14:33:48Z","snapshot_observed_at":"2026-08-09T14:50:58.651937Z","submitted_at":"2025-10-24T22:21:08Z","title":"A Vision Toward Energy-Efficient Domain-Specific Artificial Intelligence Models and Agents","version":2},"reference_index":152,"source":"pdf_text","source_observed_at":"2026-08-04T08:12:24.535811Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2510.22052"},"observation_digest":"sha256:d75bf9289d83b97674e97f0193707b5d6b102f85ef8d7bf02581fd69b99ed3f3","observation_id":"114051dc-82fc-4956-ad02-29751a4ec932","resolution":{"observed_at":"2026-08-04T08:12:24.535811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-04T07:36:36.566867Z","title":"arXiv:2112.06905","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2510.25824","last_updated":"2026-07-27T20:50:55Z","snapshot_observed_at":"2026-08-07T12:03:29.706461Z","submitted_at":"2025-10-29T18:00:00Z","title":"The Ray Tracing Sampler: Bayesian Sampling of Neural Networks for Everyone","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T07:36:36.566867Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2510.25824"},"observation_digest":"sha256:6b17d302629f73e05a1251a4166c634a8f42997683bae94f1b7ee8787b5a3ca7","observation_id":"9a54603f-628a-4d12-934c-30c4f2595c77","resolution":{"observed_at":"2026-08-04T07:36:36.566867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-08T23:43:56.161397Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T03:29:16.555166Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:505b280d5684c4aba3770f009eea7cd46ec605ea031dd621966ea122836211fc","observation_id":"d97d9803-7b68-4eac-afc3-3642c2f172cd","resolution":{"observed_at":"2026-05-10T03:29:21.519092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-08T23:43:56.161397Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T02:03:02.654035Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:7bffe50620f5030aa80a045d244b04b1033337dee7fac8233796ca06c8b327f0","observation_id":"02157ec7-66e4-4b20-9e39-19a3021f6297","resolution":{"observed_at":"2026-05-12T02:06:15.356374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2605.02300","last_updated":"2026-05-04T07:48:02Z","snapshot_observed_at":"2026-07-06T23:15:23.301944Z","submitted_at":"2026-05-04T07:48:02Z","title":"A Meta Reinforcement Learning Approach to Goals-Based Wealth Management","version":1},"reference_index":262,"source":"arxiv_source","source_observed_at":"2026-05-08T18:42:50.962120Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2605.02300"},"observation_digest":"sha256:28ba624918b975a65af48418a80ecdb2e507fffbade37a40c9cc4722bf4228c1","observation_id":"36039638-87a5-4c8d-bc05-45dd6f603b3e","resolution":{"observed_at":"2026-05-08T18:44:01.879605Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2605.11800","last_updated":"2026-05-12T08:57:59Z","snapshot_observed_at":"2026-07-06T23:23:34.924221Z","submitted_at":"2026-05-12T08:57:59Z","title":"ROMER: Expert Replacement and Router Calibration for Robust MoE LLMs on Analog Compute-in-Memory Systems","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-13T07:26:14.798902Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2605.11800"},"observation_digest":"sha256:fae15b173e2a6061fc82af8330e3403edc2161595a453b3691abc934e8bb849a","observation_id":"a53b1ea1-3b78-4027-ad55-e6f0b6507373","resolution":{"observed_at":"2026-05-13T07:27:29.565438Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2606.00275","last_updated":"2026-05-29T19:08:20Z","snapshot_observed_at":"2026-08-08T17:56:30.968996Z","submitted_at":"2026-05-29T19:08:20Z","title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T22:43:33.929871Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2606.00275"},"observation_digest":"sha256:11c613d5801322cd78dba8b85c30d7cc6f99d77050e61d4bd74b3b25e3520e18","observation_id":"24775fef-9c92-449c-acfc-60660f40295c","resolution":{"observed_at":"2026-07-01T19:26:00.058669Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":"2112.06905","doi":"10.48550/arxiv.2112.06905","metadata_source":"arxiv_reference","pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2112.06905 , year =","venue":"arXiv (Cornell University)","work_id":"34546f42-6cd7-499f-8af0-57d965de5408","year":2022},"citing_paper":{"arxiv_id":"2606.22541","last_updated":"2026-06-21T14:57:45Z","snapshot_observed_at":"2026-08-05T19:10:35.929740Z","submitted_at":"2026-06-21T14:57:45Z","title":"ASAP: A Disaggregated and Asynchronous Inference System for MoE Prefill","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-26T09:42:39.573568Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2606.22541"},"observation_digest":"sha256:9c3498805f16bf51fa514b5c5bc17608532bd67d359315d1ab09e7fa11b48925","observation_id":"3e2be5df-f02b-4fcb-84cb-790b57ddc27c","resolution":{"observed_at":"2026-07-04T09:39:46.809933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06905","snapshot_observed_at":"2026-08-01T02:37:54.336046Z","title":"Glam: Efficient scaling of language models with mixture-of-experts,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.25380","last_updated":"2026-07-28T07:34:53Z","snapshot_observed_at":"2026-08-08T07:59:40.180280Z","submitted_at":"2026-07-28T07:34:53Z","title":"Memory for Large Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-01T02:37:54.336046Z"},"links":{"cited_paper":"/paper/2112.06905","citing_paper":"/paper/2607.25380"},"observation_digest":"sha256:4892f94f9226a3d5706c2bcf792c63d8522ccea1207945dd17a768f415f8b39f","observation_id":"67c4ef99-73a8-42ff-b43a-b2b7265ea495","resolution":{"observed_at":"2026-08-01T02:37:54.336046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2112.06905/citation-record","integrity":"/paper/2112.06905/integrity","json":"/paper/2112.06905/citation-record.json","paper":"/paper/2112.06905"},"outbound":[],"paper":{"arxiv_id":"2112.06905","last_updated":"2022-08-01T21:07:58Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T12:18:14.988251Z","submitted_at":"2021-12-13T18:58:19Z","title":"GLaM: Efficient Scaling of Language Models with Mixture-of-Experts"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2112.06905."}