{"as_of":"2026-08-07T18:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:51d66d7d79ee873f1c43c08abd7f9cad85fe8920cc71e4d2e9f4f1b93ad4e20d","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T19:17:59.044982Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T07:54:47.738345Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.04422","snapshot_observed_at":"2026-08-02T07:54:47.738345Z","title":"arXiv preprint arXiv:2607.04422 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.09800","last_updated":"2026-07-28T02:32:08Z","snapshot_observed_at":"2026-08-02T07:54:44.716900Z","submitted_at":"2026-07-09T15:23:42Z","title":"Reference Traces for Auditing Invisible Weight Updates and Guiding Exact-Budget Protection","version":3},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-02T07:54:47.738345Z"},"links":{"cited_paper":"/paper/2607.04422","citing_paper":"/paper/2607.09800"},"observation_digest":"sha256:6605728e9f5214534333bc6b77fae9b3a73c57819532cd5570f2bcf1a7a6c58a","observation_id":"dd01480b-b33a-40e7-a3e2-bf042a043c86","resolution":{"observed_at":"2026-08-02T07:54:47.738345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2607.04422/citation-record","integrity":"/paper/2607.04422/integrity","json":"/paper/2607.04422/citation-record.json","paper":"/paper/2607.04422"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Pretraining large language models with nvfp4.arXiv preprint arXiv:2509.25149, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:e9d86d6bdc0a4862a29904c5bb3bccfe9bbda0794df0e8d20b129a60f3fe27a1","observation_id":"65b4cc52-d890-40c6-aeeb-5c56bf7879d1","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19115","last_updated":"2025-08-10T07:10:29Z","snapshot_observed_at":"2026-08-07T14:17:55.957254Z","submitted_at":"2025-05-25T12:14:25Z","title":"FP4 All the Way: Fully Quantized Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19115","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Fp4 all the way: Fully quantized training of llms.arXiv preprint arXiv:2505.19115, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2505.19115","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:e9ca5ff1d8d69555620f9d97b317412f6a73f3eae8cd4bded19ad92e692ac204","observation_id":"5eedd8b1-b69f-4922-9e04-185ae5971f29","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Quartet: Native fp4 training can be optimal for large language models.arXiv preprint arXiv:2505.14669, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:a8ce13df41663a2632e44e46c2b9f825a4611587ee1650ea561f0e0dd1ae4952","observation_id":"a99b5e2d-9dba-43e2-b6a3-0c365219b1a3","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Dissecting outlier dynamics in llm nvfp4 pretraining.arXiv preprint arXiv:2602.02047, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:c6d0c5e851a8e9a8fdc948bb1008516e34a48c458d3f8c46fe0fc7f01d2ef25e","observation_id":"dffadbd0-1854-48cf-a090-cb6f8beef9c5","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02010","last_updated":"2026-05-09T05:39:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-01T18:59:45Z","title":"Four Over Six: More Accurate NVFP4 Quantization with Adaptive Block Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.02010","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Four over six: More accurate nvfp4 quantization with adaptive block scaling.arXiv preprint arXiv:2512.02010, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2512.02010","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:18fc49716b0642fb38669dcce8f6b6d0e35ebc2e9925d95e3848b7183121a4b2","observation_id":"2769a908-fc6f-4090-a6d5-de49e1ea5101","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20853","last_updated":"2025-07-09T03:07:28Z","snapshot_observed_at":"2026-08-07T17:40:28.516536Z","submitted_at":"2025-02-28T08:51:55Z","title":"Oscillation-Reduced MXFP4 Training for Vision Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.20853","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Oscillation-reduced mxfp4 training for vision transformers.arXiv preprint arXiv:2502.20853, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2502.20853","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:23923438125907521e2ce99952e502a965fc2eab5a2b9d3002e10fe278eb2b13","observation_id":"2f1d7d13-1cc6-4bf0-9599-bfa8ccffa5f2","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.27527","last_updated":"2026-05-11T08:14:26Z","snapshot_observed_at":"2026-07-06T22:34:33.419178Z","submitted_at":"2025-10-31T14:57:16Z","title":"TetraJet-v2: Accurate NVFP4 Training for Large Language Models with Oscillation Suppression and Outlier Control","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.27527","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Tetrajet-v2: Accurate nvfp4 training for large language models with oscillation suppression and outlier control.arXiv preprint arXiv:2510.27527, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2510.27527","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:f74cf1b1f9a1cf923d7aa96b596ec67805df3c353391a324262d76ae4148eee9","observation_id":"8a4fd6f2-b7e6-4190-ab69-55c3bbc91a83","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Muon: An optimizer for hidden layers in neural networks, 2024.URL https://kellerjordan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:6716df5af5e89c34da0401d8d77325b39456b3036886669f05897129c5baf96e","observation_id":"65ef24b8-2162-409c-ac6a-ae944a910e1e","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16982","last_updated":"2025-02-24T09:12:29Z","snapshot_observed_at":"2026-08-02T00:32:51.000665Z","submitted_at":"2025-02-24T09:12:29Z","title":"Muon is Scalable for LLM Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.16982","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Muon is scalable for llm training, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2502.16982","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:0eb678c5e2a4dcef812d1e26b0c075e28677c0595011a12b6944306a270964fd","observation_id":"b5126e53-ea5b-4606-b6dc-41b2084c5b5e","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Deepseek-v4: Towards highly efficient million-token context intelligence, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:c11bb4d8b9b53aeb45485511191346465e9bc85913409426ada98467c40de6a3","observation_id":"0aad755f-888b-4fdc-b8ad-01d0667bc3ef","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Achieving low-bit muon through subspace preservation and grid quantization","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:0a46a8da75efc52acdad94c0cbecbb0713d8ac9692804404cc717805113cc99d","observation_id":"a6835d34-790e-4fd4-aaf9-d1a499eb48d8","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19313","last_updated":"2025-02-12T23:37:50Z","snapshot_observed_at":"2026-08-07T16:34:25.679063Z","submitted_at":"2024-10-25T05:59:30Z","title":"COAT: Compressing Optimizer states and Activation for Memory-Efficient FP8 Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19313","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Coat: Compressing optimizer states and activation for memory-efficient fp8 training.arXiv preprint arXiv:2410.19313, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2410.19313","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:867d9996f0ede4c3df88b159cc4718deee7bb946c8a6df2625856bab3fe48099","observation_id":"c97efd93-6248-4447-8d7c-0cb980062dcc","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Metis: Training llms with fp4 quantization.arXiv preprint arXiv:2509.00404, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:ba6e1880cf92ea3b6794a60e18c80b9c1cfa5d2bdaeef5ea6440cb4428579f9a","observation_id":"76ea06bc-d00c-4ba5-8881-2b28fc0bfb6b","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Attn-qat: 4-bit attention with quantization-aware training.arXiv preprint arXiv:2603.00040, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:3397b24f2367e449e33705115f0b68961877967c9667d0e6afb96be4b46c5516","observation_id":"ce1da27d-e3c0-492a-8234-fe20ed290345","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Sageattention3: Microscaling fp4 attention for inference and an exploration of 8-bit training.arXiv preprint arXiv:2505.11594, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:a730c0f69340f0cc6242723c2652a1fc9b9326ab497e87b4ebe4ff6a73e55b5a","observation_id":"16c2fa9e-1c31-4668-8a90-5e9e1d80248f","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-07-06T16:33:55.369704Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Microscaling data formats for deep learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:64c9e3dedeb011fc3ae7c3a593618168f3fef44720c2fb58a3850d9bb0676f60","observation_id":"024a417c-01ec-4800-92ba-301d29e9da18","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Root: Robust orthogonalized optimizer for neural network training.arXiv preprint arXiv:2511.20626, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:a77e072176f3aa1580b651948a07d9fb89cd1b34b4f1b5dccf99d45b1b63fbdf","observation_id":"cc3a4a6b-d7e6-4fc0-9f54-c13330b1defd","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Flashattention: Fast and memory-efficient exact attention with io-awareness.Advances in neural information processing systems, 35:16344–16359, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:df068f16dad7da46c9911cf434d8e98e6588c1a4d1c15020a6f1fd29d7baedd3","observation_id":"f7fe30bb-a00a-4b95-b865-ecdf7acda074","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Flashattention-2: Faster attention with better parallelism and work partitioning.arXiv preprint arXiv:2307.08691, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:d4803d84e9f897d3c5c7d8aaa1263203b008fc0a241df5d2c46a2bbfdf0977bf","observation_id":"03d758b3-badb-4f9e-a64b-49d73b40d7aa","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20586","last_updated":"2025-08-26T23:08:09Z","snapshot_observed_at":"2026-08-07T17:41:27.066746Z","submitted_at":"2025-02-27T23:01:31Z","title":"Training LLMs with MXFP4","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.20586","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Training llms with mxfp4.arXiv preprint arXiv:2502.20586, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2502.20586","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:ab1a0181f622592e06ee6a67eb936b6736872ba3346fe7b4b1fa8460a87444e5","observation_id":"1738a3a2-3203-4f81-bbd8-8603df924636","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Svdquant: Absorbing outliers by low-rank components for 4-bit diffusion models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:7b4bfeb9199d3cc2a04479de04b0c5d1d379ad135777bf96a1ea49f8fbda31dd","observation_id":"c7d18e4a-1378-4cfe-b4b7-a315d3b3bcf1","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Finding structure with randomness: Probabilistic algorithms for constructing approximate matrix decompositions.SIAM review, 53 (2):217–288, 2011","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:cdeb084c66e9f3c8abb3a26a162f2e3ed0af7946e3c555a1cf1937b0e3a66dd5","observation_id":"650622c4-0d22-4628-b157-63549d6046ea","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.14444","last_updated":"2025-09-02T16:12:36Z","snapshot_observed_at":"2026-07-06T22:15:27.120497Z","submitted_at":"2025-08-20T06:00:57Z","title":"NVIDIA Nemotron Nano 2: An Accurate and Efficient Hybrid Mamba-Transformer Reasoning Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.14444","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Nvidia nemotron nano 2: An accurate and efficient hybrid mamba-transformer reasoning model, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2508.14444","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:aa8caa436c014ba70d7d8fe36364a593fbd8695c14fdbebb85779a5639aa405c","observation_id":"933fef17-26ff-4e69-97d7-030e7e2ccaac","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Training and inference with integers in deep neural networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:7302f9c3987797a064983b4dcf60744666737ce50ec1594e2f229bb3bbb6ba78","observation_id":"887afe75-4300-4d96-b5b4-b539a9052f90","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.05433","last_updated":"2022-09-29T20:47:07Z","snapshot_observed_at":"2026-07-06T13:51:20.183063Z","submitted_at":"2022-09-12T17:39:55Z","title":"FP8 Formats for Deep Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.05433","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Fp8 formats for deep learning.arXiv preprint arXiv:2209.05433, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2209.05433","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:6ae6cee3556b49f87c0be078cb3e0c80807a60456e8ce8ec3ee0303c338af2e9","observation_id":"7a3d4e70-c129-44b3-808f-0f95ebaca6b3","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Stable and low-precision training for large-scale vision-language models.Advances in Neural Information Processing Systems, 36:10271–10298, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:4665ab243659db5231cefa53e5966dcd9508e0c5836c9d9a8c7648e01b0de33f","observation_id":"dd200e99-2ae9-4764-8e63-308886d09899","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12517","last_updated":"2025-02-10T09:37:59Z","snapshot_observed_at":"2026-08-04T18:48:38.871823Z","submitted_at":"2024-09-19T07:15:58Z","title":"Scaling FP8 training to trillion-token LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12517","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Scaling fp8 training to trillion-token llms.arXiv preprint arXiv:2409.12517, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2409.12517","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:363eac28b837e020688bb42cabeb62bdea2b8f9357c6f7a7f57ede06b545f0bc","observation_id":"7ae5b301-63ae-49e1-8d4e-0c59f242f88d","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Deepseek-v3 technical report.arXiv preprint arXiv:2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:302ca9696cb85066482becbe58f8768f2e335d3fa72411fc39bb8246e3f4d936","observation_id":"e631cf3e-e3d2-4b17-9a83-f88810c66b60","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Quarot: Outlier-free 4-bit inference in rotated llms.Advances in Neural Information Processing Systems, 37:100213–100240, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:ac0200027c24738a51016b77ba5e7faf2471d49a6bf3cb3af284913a7a18e280","observation_id":"51b48e9c-9ca3-4ca6-ad5c-940bf0d1d5e8","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17116","last_updated":"2025-05-23T09:44:25Z","snapshot_observed_at":"2026-08-02T15:17:06.428629Z","submitted_at":"2025-01-28T18:04:50Z","title":"Optimizing Large Language Model Training Using FP4 Quantization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17116","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Optimizing large language model training using fp4 quantization.arXiv preprint arXiv:2501.17116, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2501.17116","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:dc794f976ee7284530db476e3da02a6f93ea56d8af03625d10996312e65e3084","observation_id":"3be8d1c5-0b54-4f28-bbec-c9ef66a52809","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.02861","last_updated":"2022-06-20T16:05:15Z","snapshot_observed_at":"2026-07-06T11:55:04.054344Z","submitted_at":"2021-10-06T15:43:20Z","title":"8-bit Optimizers via Block-wise Quantization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.02861","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"8-bit optimizers via block-wise quantization.arXiv preprint arXiv:2110.02861, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2110.02861","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:90b4b762afbe16c5c16894ce0e20e57eb482db8e9c2cbf8022858bdeb3a80576","observation_id":"0814f2e2-e831-447a-8c90-bd2e633f1f36","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Memory efficient optimizers with 4-bit states.Advances in Neural Information Processing Systems, 36:15136–15171, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:acbda7f70166e5f9965543f15f7383aa753a973fa9438c6fa653d758df7d9fbf","observation_id":"2ba6c902-29f3-4a49-8a28-45ee412a64a5","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Effective quantization of muon optimizer states.arXiv preprint arXiv:2509.23106, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:09c0a2ce1239ea457598900ecd640d964950843bbdeb8eeb15b4bb43f5219125","observation_id":"bdd81a36-5f49-4ab0-8405-cda1554772e6","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Sageat- tention: Accurate 8-bit attention for plug-and-play inference acceleration.arXiv preprint arXiv:2410.02367, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:de68036afd463e4bed50f3055e8096c44c43270d256faa87022da9fd61bc5461","observation_id":"80a5171d-09fe-418f-be6c-8cc164953c46","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Sageat- tention2: Efficient attention with thorough outlier smoothing and per-thread int4 quantization","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:03b866e9d7bc212bdadbded7fb55273441ab2bc438b9f8fbc90102c7f0e2ee59","observation_id":"b97251a2-4038-489e-aebc-7aae3b804ccc","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"variance","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:c45d346e71168bfc6033aa6c2d5cbe76b731090dff2b9a5153c60c5b3e8822d4","observation_id":"957beb04-0120-4da7-8ac8-a7848030d5d1","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.04422","last_updated":"2026-07-05T17:31:36Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-04T03:31:46.106169Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":36,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 1 inbound Pith citation observation for arXiv:2607.04422."}