{"as_of":"2026-08-08T23:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7dc2bf76fda1907f26508c1ada4142fd965ad1d7d445e2bcb5feb9a0ac6e10fc","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":23,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":23,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":23,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:20:23.627982Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":12,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-11T06:53:58.960357Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2305.13245"},"observation_digest":"sha256:cf2619b3146feeb6a0fe476264ca456200e1d529619100a2d9dadff666157aed","observation_id":"67431689-0637-42ec-a05c-6925b5ded578","resolution":{"observed_at":"2026-05-11T06:53:59.369167Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2401.15947","last_updated":"2024-12-23T08:05:14Z","snapshot_observed_at":"2026-08-06T02:31:58.372974Z","submitted_at":"2024-01-29T08:13:40Z","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T02:33:30.143907Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2401.15947"},"observation_digest":"sha256:7a64fe446b3ce2e3619897309da99b68511965aea21c3b0369dc9967b7e218a2","observation_id":"df4dd422-f982-4b2e-a421-ca6db158304c","resolution":{"observed_at":"2026-05-16T02:33:30.309336Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T18:00:53.389420Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2404.06395"},"observation_digest":"sha256:fad3ec3a56c216110ec95509de238d11a8a891edb287579ceb7a9dbb4da978a5","observation_id":"56704cea-9f6c-4252-81d6-88872d84893d","resolution":{"observed_at":"2026-05-13T18:00:53.537329Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:ec19e254ddf4e06a1c73863dce71e9c2e712e8969de740750671a2ee3e142102","observation_id":"c041c756-97d8-414c-af0e-3c4532ca826f","resolution":{"observed_at":"2026-05-15T02:39:33.232205Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2502.04416","last_updated":"2026-04-23T00:51:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-06T14:05:30Z","title":"Analytical FFN-to-MoE Restructuring via Activation Pattern Analysis","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-23T04:08:29.089438Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2502.04416"},"observation_digest":"sha256:7603d3b89f80befaee23ec3ca665be438ba1ea5c063bb85c90853ae095105da0","observation_id":"6f0a874f-b08c-46e0-93b3-a2d5562c126d","resolution":{"observed_at":"2026-05-23T04:12:31.096197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-08T11:20:23.627982Z","title":"Krajewski, J., Ludziejewski, J., Adamczewski, K., Pi ´oro, M., Krutul, M., Antoniak, S., Ciebiera, K., Kr ´ol, K., Odrzyg´o´zd´z, T., Sankowski, P., Cygan, M., and Jaszczur, S","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.07972","last_updated":"2025-03-09T19:39:00Z","snapshot_observed_at":"2026-08-08T11:14:28.908181Z","submitted_at":"2025-02-11T21:36:31Z","title":"Training Sparse Mixture Of Experts Text Embedding Models","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-08T11:20:23.627982Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2502.07972"},"observation_digest":"sha256:710c76470f8d983615540788d8cddd8ebf9d0f6a804a51efe32573e9790f44fa","observation_id":"04f50ded-00a7-460e-935f-a8e016755d0e","resolution":{"observed_at":"2026-08-08T11:20:23.627982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2506.12119","last_updated":"2026-05-17T08:18:55Z","snapshot_observed_at":"2026-08-02T23:22:23.397250Z","submitted_at":"2025-06-13T17:59:05Z","title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T00:05:08.916339Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2506.12119"},"observation_digest":"sha256:82eb0c615022b5f281c1446fadf6ce31f8a374535f207398bd095085541a12b8","observation_id":"add9a803-4985-4aa2-9871-2832581282d2","resolution":{"observed_at":"2026-05-22T00:05:47.760189Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2509.05276","last_updated":"2026-05-08T09:41:26Z","snapshot_observed_at":"2026-07-06T22:24:38.247543Z","submitted_at":"2025-09-05T17:34:00Z","title":"SpikingBrain: Spiking Brain-inspired Large Models","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-18T18:51:06.243305Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2509.05276"},"observation_digest":"sha256:ede6cd142b3bad7a9210848c5db225664cbc9242544a78e917f7922655e646ac","observation_id":"277012d9-716a-4bd7-abd1-de2f81775792","resolution":{"observed_at":"2026-05-18T18:51:45.676873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-04T14:41:38.609117Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T03:29:16.555166Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:8dab70cebc5cd3eddded652069941d311a646385c6288c0ec86a4c1c66255d03","observation_id":"243a998b-784f-4ebc-ae15-bca7d1e43c11","resolution":{"observed_at":"2026-05-10T03:29:21.454800Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2604.19835","last_updated":"2026-05-10T18:33:52Z","snapshot_observed_at":"2026-08-04T14:41:38.609117Z","submitted_at":"2026-04-21T05:53:33Z","title":"Expert Upcycling: Shifting the Compute-Efficient Frontier of Mixture-of-Experts","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T02:03:02.654035Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2604.19835"},"observation_digest":"sha256:fcdd1b05a8925a680dfd29632fd200dd85a95aba1d62ff10248c5394dcc93466","observation_id":"14ed15c8-2a8e-4c1f-944b-0fa2b58d5e0f","resolution":{"observed_at":"2026-05-12T02:06:15.330760Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2605.13997","last_updated":"2026-05-13T18:07:12Z","snapshot_observed_at":"2026-07-06T23:25:29.856145Z","submitted_at":"2026-05-13T18:07:12Z","title":"HodgeCover: Higher-Order Topological Coverage Drives Compression of Sparse Mixture-of-Experts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-15T05:54:32.496951Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2605.13997"},"observation_digest":"sha256:f9ca46eb3c08a60d78db934ceae777906d54ea54d196952b18ceedeb64f0f241","observation_id":"2cdf9ed4-ec0b-4c69-aeb3-2717b7a78f4c","resolution":{"observed_at":"2026-05-15T05:55:04.804235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2605.26496","last_updated":"2026-05-26T03:19:04Z","snapshot_observed_at":"2026-07-06T23:36:20.072253Z","submitted_at":"2026-05-26T03:19:04Z","title":"Dense2MoE: Pushing the Pareto Frontier of On-Device LLMs via Unified Pruning and Upcycling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T19:34:17.270161Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2605.26496"},"observation_digest":"sha256:f18acd8b82f41884f642463d0e7fde36e38ed24241d92b1408f0fa3d73604ac7","observation_id":"411bc5a4-1c53-437d-a038-161510294170","resolution":{"observed_at":"2026-06-29T19:43:55.024776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.00275","last_updated":"2026-05-29T19:08:20Z","snapshot_observed_at":"2026-08-08T17:56:30.968996Z","submitted_at":"2026-05-29T19:08:20Z","title":"Hyperbolic and Evidence-Prioritized Experts for Large Vision-Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T22:43:33.929871Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.00275"},"observation_digest":"sha256:54e0ab76da74cd1e7f960f8e05895438a5718b0521902d6b8cbbe723dd8c5709","observation_id":"9b7f1f9f-bc4c-4b2f-94ae-612a300fab00","resolution":{"observed_at":"2026-07-01T19:26:00.040544Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.07404","last_updated":"2026-06-05T15:48:42Z","snapshot_observed_at":"2026-07-06T23:47:05.104295Z","submitted_at":"2026-06-05T15:48:42Z","title":"Reversible Foundations: Training a 120B Sparse MoE through State-Preserving Scaling","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-06-27T22:55:09.477413Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.07404"},"observation_digest":"sha256:5cccd600dedb218d66c9e2f45748d5b10850757310147a49ed3ff9ffd331ef49","observation_id":"7b902ecc-0357-479a-8112-0b8d402e17ee","resolution":{"observed_at":"2026-07-02T16:07:09.420797Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.09038","last_updated":"2026-06-08T05:10:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-06-08T05:10:05Z","title":"Personalization Meets Safety:Mechanisms,Risks,and Mitigations in Personalized LLMs","version":1},"reference_index":197,"source":"pdf_text","source_observed_at":"2026-06-27T16:49:14.243931Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.09038"},"observation_digest":"sha256:4131482930cbf6c734b77828331d738f876774f9c597e67491eb8fcd918d7be2","observation_id":"a217183e-a6fe-4e15-8af5-82bc0caf9123","resolution":{"observed_at":"2026-07-03T01:07:30.270932Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.10369","last_updated":"2026-06-09T03:28:17Z","snapshot_observed_at":"2026-08-06T17:51:41.080624Z","submitted_at":"2026-06-09T03:28:17Z","title":"PADD: Path-Aligned Decompression Distillation for Non-Router Teacher to Guide MoE Student Learning","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-27T13:18:41.596372Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.10369"},"observation_digest":"sha256:e132de0dcea76a7daea33c9f0ba1d5f7ffb6a0b8556cd960e6c0c3ddf70923d7","observation_id":"165e8b90-51e0-4b77-b0be-2e47efa18e5a","resolution":{"observed_at":"2026-07-03T05:27:39.751050Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.21645","last_updated":"2026-06-19T17:56:44Z","snapshot_observed_at":"2026-08-05T10:02:01.473600Z","submitted_at":"2026-06-19T17:56:44Z","title":"Behavioral and Representational Evidence of Binomial Ordering Preferences in Large Language Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-26T14:18:11.215278Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.21645"},"observation_digest":"sha256:6162610f3e67fc97af10b74a725faf5113f2fc69274ae2eee43c8d8f252aa886","observation_id":"fec9ca77-1ce8-4142-b12f-91919d556491","resolution":{"observed_at":"2026-07-04T06:39:37.844241Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.24901","last_updated":"2026-06-12T13:44:48Z","snapshot_observed_at":"2026-08-03T03:21:51.293576Z","submitted_at":"2026-06-12T13:44:48Z","title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T05:02:18.347642Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.24901"},"observation_digest":"sha256:5d661de29aaa3b574565bc90770c80c15a61db17dea969850e1551ba7c7df6a8","observation_id":"e307e6d8-b21f-4c43-8e96-2d76a4f0c5ad","resolution":{"observed_at":"2026-07-03T16:48:39.909349Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2606.26287","last_updated":"2026-06-24T18:34:00Z","snapshot_observed_at":"2026-07-07T00:00:36.766142Z","submitted_at":"2026-06-24T18:34:00Z","title":"GeMoE: Gating Entropy is All You Need for Uncertainty-aware Adaptive Routing in MoE-based Large Vision-Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-26T01:32:40.435742Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2606.26287"},"observation_digest":"sha256:7237efc60830310c1d80d6fe07e7147e8f39907c946fc67aaf86c246c4d4019f","observation_id":"f82ce639-7bc7-41db-af5d-7f85c2da1d68","resolution":{"observed_at":"2026-07-04T15:39:56.508142Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":"2212.05055","doi":"10.48550/arxiv.2212.05055","metadata_source":"arxiv_reference","pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":"arXiv (Cornell University)","work_id":"9ce112e7-e6f9-4837-9a98-f630f620a73b","year":2022},"citing_paper":{"arxiv_id":"2607.00293","last_updated":"2026-07-01T00:42:40Z","snapshot_observed_at":"2026-08-05T17:01:51.209148Z","submitted_at":"2026-07-01T00:42:40Z","title":"Rosetta: Composable Native Multimodal Pretraining","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-02T15:41:13.556849Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2607.00293"},"observation_digest":"sha256:63a0b5561f9c72e466ef89ab19a21a2f904f4da1b09e4a16f4c4f0c3da2c9495","observation_id":"bd9b0fee-c622-4e64-b79e-09f147b71b1a","resolution":{"observed_at":"2026-07-02T15:47:05.759103Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-07-11T19:16:57.396710Z","title":"Sparse upcycling: Training mixture-of-experts from dense checkpoints","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.04426","last_updated":"2026-07-05T17:43:06Z","snapshot_observed_at":"2026-08-02T16:12:54.212857Z","submitted_at":"2026-07-05T17:43:06Z","title":"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI","version":1},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-07-11T19:16:57.396710Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2607.04426"},"observation_digest":"sha256:ede0a9c8464fe15f331fbdf172ceb4e262f0c2b750e62d5ed7ad240241f04284","observation_id":"9f0979a5-30ab-4193-a1fd-ce42ab786fdc","resolution":{"observed_at":"2026-07-11T19:16:57.396710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-02T11:58:30.699815Z","title":"Sparse Upcycling : Training Mixture-of-Experts from Dense Checkpoints","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22586","last_updated":"2026-06-09T04:56:59Z","snapshot_observed_at":"2026-08-06T10:04:16.098099Z","submitted_at":"2026-06-09T04:56:59Z","title":"MM-ShiftKV: Decode-Aware Prefill-Stage KV Selection for Multimodal Large Language Models","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-02T11:58:30.699815Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2607.22586"},"observation_digest":"sha256:a6fc97107f9d1a2845fa8106183b4b83dbb2b4a11743f6606d2d8696d64cd86c","observation_id":"b7d7e243-73ff-4194-a052-a263e3d8d0a1","resolution":{"observed_at":"2026-08-02T11:58:30.699815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.05055","snapshot_observed_at":"2026-08-02T10:18:55.487992Z","title":"arXiv:2212.05055","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24787","last_updated":"2026-07-30T11:13:04Z","snapshot_observed_at":"2026-08-08T07:03:33.662559Z","submitted_at":"2026-06-24T04:53:00Z","title":"SpecPrefetch: Parameter-Efficient Expert Prefetching for Sparse MoE Foundation Models","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T10:18:55.487992Z"},"links":{"cited_paper":"/paper/2212.05055","citing_paper":"/paper/2607.24787"},"observation_digest":"sha256:67aee2b5ec8a97318aeeb8fd3e896da2cf7b145b03fcb7a1842155af9b21e0fb","observation_id":"8df7d279-6abf-4491-a22b-8809a8729a3d","resolution":{"observed_at":"2026-08-02T10:18:55.487992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2212.05055/citation-record","integrity":"/paper/2212.05055/integrity","json":"/paper/2212.05055/citation-record.json","paper":"/paper/2212.05055"},"outbound":[],"paper":{"arxiv_id":"2212.05055","last_updated":"2023-02-17T17:54:50Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T14:28:52.486524Z","submitted_at":"2022-12-09T18:57:37Z","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 23 inbound Pith citation observations for arXiv:2212.05055."}