{"as_of":"2026-08-20T07:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:bb7c115b260e644bca11d32d47e9f983c7c1fff3ee39a9cd0de92dae3e16a050","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":78,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":78,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":78,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":78,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:23:53.379727Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-12T20:16:46.702310Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.09909","last_updated":"2025-05-30T01:11:39Z","snapshot_observed_at":"2026-08-16T04:31:40.072568Z","submitted_at":"2024-11-15T03:11:19Z","title":"AMXFP4: Taming Activation Outliers with Asymmetric Microscaling Floating-Point for 4-bit LLM Inference","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-12T20:16:46.702310Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2411.09909"},"observation_digest":"sha256:b2945c7bf0a8cd03df45b8b26b4df603b0c2b3e216c972cc70bd5627ee6ef4e5","observation_id":"959aa0db-96b4-4463-baca-1b0164f3a122","resolution":{"observed_at":"2026-08-12T20:16:46.702310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-12T17:59:21.777415Z","title":"Microscal- ing data formats for deep learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.12090","last_updated":"2024-12-20T16:46:47Z","snapshot_observed_at":"2026-08-19T00:56:14.926689Z","submitted_at":"2024-11-18T22:01:32Z","title":"Hardware Trends Impacting Floating-Point Computations In Scientific Applications","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T17:59:21.777415Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2411.12090"},"observation_digest":"sha256:3a5a8aa1d27e0bf3753b868d667938494ba0bed40705ef18d09ca2e8eae84aaf","observation_id":"e94ecf9b-a75d-4f99-8961-da59893d5af7","resolution":{"observed_at":"2026-08-12T17:59:21.777415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-12T16:43:50.601037Z","title":"Microscaling data formats for deep le arning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13259","last_updated":"2024-11-20T12:20:45Z","snapshot_observed_at":"2026-08-18T14:12:49.024949Z","submitted_at":"2024-11-20T12:20:45Z","title":"Interface for Sparse Linear Algebra Operations","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T16:43:50.601037Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2411.13259"},"observation_digest":"sha256:a03dfefc9c89d8d9f0365aa9e1300dbd1ef75fec2c8666a4c0ebaa5acede983b","observation_id":"324db509-fa19-433b-98d8-ff586306e3f7","resolution":{"observed_at":"2026-08-12T16:43:50.601037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-12T11:37:12.587689Z","title":"arXiv:2310.10537 [cs.LG] https://arxiv.org/abs/2310.10537","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.18065","last_updated":"2025-06-04T01:25:53Z","snapshot_observed_at":"2026-08-18T10:18:08.712278Z","submitted_at":"2024-11-27T05:16:26Z","title":"FlexiBit: Fully Flexible Precision Bit-parallel Accelerator Architecture for Arbitrary Mixed Precision AI","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T11:37:12.587689Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2411.18065"},"observation_digest":"sha256:afa3ee09bd02c06a94da453af77593611805840ed54dc888a79cccfd6059e807","observation_id":"f3aca87b-e858-4abc-a9a4-d958f173d0ee","resolution":{"observed_at":"2026-08-12T11:37:12.587689Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-11T21:11:39.299906Z","title":"D., Zhao, R., More, A., Hall, M., Khodamoradi, A., Deng, S., Choudhary, D., Cornea, M., Dellinger, E., Denolf, K., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.04964","last_updated":"2024-12-11T13:27:00Z","snapshot_observed_at":"2026-08-18T12:58:44.217063Z","submitted_at":"2024-12-06T11:29:32Z","title":"Flash Communication: Reducing Tensor Parallelization Bottleneck for Fast Large Language Model Inference","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-11T21:11:39.299906Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2412.04964"},"observation_digest":"sha256:10383ae862609f174d02507b82fc82ac3609c7fdb4bd3b055e2048dd6643292f","observation_id":"c3c6efde-3bb9-4f8f-b95c-5fb0b86ad6b8","resolution":{"observed_at":"2026-08-11T21:11:39.299906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-10T22:44:30.157115Z","title":"D., Zhao, R., Elango, V ., Shafipour, R., Hall, M., Mesmakhosroshahi, M., More, A., Melnick, L., Golub, M., Varatkar, G., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.01144","last_updated":"2025-07-24T03:46:03Z","snapshot_observed_at":"2026-08-18T02:48:06.805374Z","submitted_at":"2025-01-02T08:57:00Z","title":"BlockDialect: Block-wise Fine-grained Mixed Format Quantization for Energy-Efficient LLM Inference","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T22:44:30.157115Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2501.01144"},"observation_digest":"sha256:45f3eb9355f77740f64aa22ff0774d0295a6aadc950fb3c1196c04517993ec60","observation_id":"30a922f6-ae9a-4391-9fc2-12d018e30109","resolution":{"observed_at":"2026-08-10T22:44:30.157115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-10T14:20:56.777219Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.15448","last_updated":"2025-01-26T08:34:26Z","snapshot_observed_at":"2026-08-18T14:12:07.338954Z","submitted_at":"2025-01-26T08:34:26Z","title":"SQ-DM: Accelerating Diffusion Models with Aggressive Quantization and Temporal Sparsity","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T14:20:56.777219Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2501.15448"},"observation_digest":"sha256:a5c79008cb4ce90a3b815636e4c40500ef185ee198c606051c8cc8cb4f973a4d","observation_id":"43f6fff7-a872-4f6f-87a4-8e47fe87f75a","resolution":{"observed_at":"2026-08-10T14:20:56.777219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-10T17:22:18.848683Z","title":"D.; Zhao, R.; More, A.; Hall, M.; Khodamoradi, A.; Deng, S.; Choudhary, D.; Cornea, M.; Dellinger, E.; Denolf, K.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00026","last_updated":"2025-02-07T12:23:59Z","snapshot_observed_at":"2026-08-20T05:38:18.660587Z","submitted_at":"2025-01-21T17:10:52Z","title":"Pushing the Limits of BFP on Narrow Precision LLM Inference","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T17:22:18.848683Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2502.00026"},"observation_digest":"sha256:02c69921fbac4c120cc7e68424a38c7a513afc4da4bf33bb871955bc0eec6c4f","observation_id":"9245f61c-c08f-4b2b-b863-9419f3c24209","resolution":{"observed_at":"2026-08-10T17:22:18.848683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-16T12:23:53.379727Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.12984","last_updated":"2025-08-31T22:12:47Z","snapshot_observed_at":"2026-08-18T14:12:07.808660Z","submitted_at":"2025-04-17T14:45:03Z","title":"Tilus: A Tile-Level GPGPU Programming Language for Low-Precision Computation","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T12:23:53.379727Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2504.12984"},"observation_digest":"sha256:54dea808370216b4c01ea86e13d7186c66d560995e845b443a2f3276c5a646ad","observation_id":"3f317212-e3a0-455a-9f01-eabbdba8f908","resolution":{"observed_at":"2026-08-16T12:23:53.379727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-16T12:00:43.119885Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.14152","last_updated":"2025-04-19T02:51:45Z","snapshot_observed_at":"2026-08-19T00:46:04.287923Z","submitted_at":"2025-04-19T02:51:45Z","title":"FGMP: Fine-Grained Mixed-Precision Weight and Activation Quantization for Hardware-Accelerated LLM Inference","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T12:00:43.119885Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2504.14152"},"observation_digest":"sha256:47a8af65f2da1afe1a60a645b68cc11a4ec490454ca1243e1d794e0b65e5cd6a","observation_id":"ddd75e2a-0e21-4ac1-8c67-ae2f0c25e74b","resolution":{"observed_at":"2026-08-16T12:00:43.119885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-16T04:30:55.947370Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.01043","last_updated":"2026-07-29T08:00:39Z","snapshot_observed_at":"2026-08-18T07:27:00.226721Z","submitted_at":"2025-05-02T06:33:25Z","title":"Low-Precision Training of Large Language Models: Methods, Challenges, and Opportunities","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T04:30:55.947370Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.01043"},"observation_digest":"sha256:729d1612cee1d630a13c55e0e23e6f7aa5851e1803c356eb1530869386e77efc","observation_id":"7f4fea8b-ac56-4d43-b213-6f240e7505d1","resolution":{"observed_at":"2026-08-16T04:30:55.947370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-15T21:03:33.107145Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.11170","last_updated":"2025-05-16T12:14:12Z","snapshot_observed_at":"2026-08-19T00:45:26.987657Z","submitted_at":"2025-05-16T12:14:12Z","title":"Gaussian Weight Sampling for Scalable, Efficient and Stable Pseudo-Quantization Training","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T21:03:33.107145Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.11170"},"observation_digest":"sha256:1dfdbd845811d2d51303dec50112e024aab900a13d8bb27640e52598bdda140d","observation_id":"6bd37c6d-4928-4491-9011-c5ec9b5cdc82","resolution":{"observed_at":"2026-08-15T21:03:33.107145Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-15T20:25:33.242465Z","title":"Available: http://arxiv.org/abs/2310.10537","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.13159","last_updated":"2025-05-19T14:18:34Z","snapshot_observed_at":"2026-08-18T14:12:08.029484Z","submitted_at":"2025-05-19T14:18:34Z","title":"MXDOTP: A RISC-V ISA Extension for Enabling Microscaling (MX) Floating-Point Dot Products","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T20:25:33.242465Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.13159"},"observation_digest":"sha256:4b6cd9a5d1d8a26303b473e5a873ba80e3a006375d75ea09874d3ce019448449","observation_id":"897470db-ef0a-4c88-a08c-aea47e2fba63","resolution":{"observed_at":"2026-08-15T20:25:33.242465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T15:41:08.331673Z","title":"Microscaling data formats for deep learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.14302","last_updated":"2025-05-20T12:54:43Z","snapshot_observed_at":"2026-08-14T10:57:46.400268Z","submitted_at":"2025-05-20T12:54:43Z","title":"Scaling Law for Quantization-Aware Training","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:41:08.331673Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.14302"},"observation_digest":"sha256:31004d796b5904a5d33eb6e584c2c451ca6d479f3ff7e172570b59820c777302","observation_id":"e0aade8b-4176-40b5-9bad-2d26c00f6f3f","resolution":{"observed_at":"2026-08-07T15:41:08.331673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T15:08:19.316349Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16335","last_updated":"2025-05-22T07:47:51Z","snapshot_observed_at":"2026-08-18T14:05:13.471702Z","submitted_at":"2025-05-22T07:47:51Z","title":"FPQVAR: Floating Point Quantization for Visual Autoregressive Model with FPGA Hardware Co-design","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:08:19.316349Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.16335"},"observation_digest":"sha256:29433325759228b072c630e5d1534cdf5a50da31babeb85ced82f209e2dd9a7e","observation_id":"889b3360-a0d1-49be-9c43-d9a2f099cc22","resolution":{"observed_at":"2026-08-07T15:08:19.316349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T15:06:30.167879Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16346","last_updated":"2025-05-23T07:21:27Z","snapshot_observed_at":"2026-08-17T05:34:15.756265Z","submitted_at":"2025-05-22T07:59:26Z","title":"How to keep pushing ML accelerator performance? Know your rooflines!","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:06:30.167879Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.16346"},"observation_digest":"sha256:36ba4424332aaeaf904c481c2f6381d3173adb7f4744a56c09b3209eeb55e2d7","observation_id":"99a00f12-8207-439c-b6f2-350163c31afb","resolution":{"observed_at":"2026-08-07T15:06:30.167879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T14:25:38.750726Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.19115","last_updated":"2025-08-10T07:10:29Z","snapshot_observed_at":"2026-08-17T14:20:08.812124Z","submitted_at":"2025-05-25T12:14:25Z","title":"FP4 All the Way: Fully Quantized Training of LLMs","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:25:38.750726Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2505.19115"},"observation_digest":"sha256:6571a6fe3210d09ceeab0e36610f24505c8602debaf114555896bf8b3b6a5884","observation_id":"7b4d5fc0-00e7-4635-bd23-bcd65a4c4de4","resolution":{"observed_at":"2026-08-07T14:25:38.750726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T10:26:32.636714Z","title":"Microscaling Data Formats for Deep Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05508","last_updated":"2025-06-05T18:47:49Z","snapshot_observed_at":"2026-08-17T14:56:44.646120Z","submitted_at":"2025-06-05T18:47:49Z","title":"Beyond the Buzz: A Pragmatic Take on Inference Disaggregation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.636714Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2506.05508"},"observation_digest":"sha256:5f492dad28db701abfd2852d028c85d9d26373c80c0611d3371d95f6a43bf729","observation_id":"b05af5ff-ab51-4ff7-b3ef-96242990f417","resolution":{"observed_at":"2026-08-07T10:26:32.636714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-07T12:12:45.790981Z","title":"Microscaling data formats for deep learning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.08027","last_updated":"2025-08-18T19:51:06Z","snapshot_observed_at":"2026-08-13T17:08:54.114895Z","submitted_at":"2025-05-30T21:08:15Z","title":"Recipes for Pre-training LLMs with MXFP8","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:45.790981Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2506.08027"},"observation_digest":"sha256:2a5b8a80004fc48df49d4130199ed0fc1fdacc7a05c9d596b0aa80d1523b38d0","observation_id":"ea954705-fd04-42aa-92d2-feba37489914","resolution":{"observed_at":"2026-08-07T12:12:45.790981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-15T19:09:44.454827Z","title":"Microscaling Data Formats for Deep Learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.17768","last_updated":"2025-06-21T17:37:22Z","snapshot_observed_at":"2026-08-18T14:12:07.537773Z","submitted_at":"2025-06-21T17:37:22Z","title":"Log-Normal Multiplicative Dynamics for Stable Low-Precision Training of Large Networks","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-15T19:09:44.454827Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2506.17768"},"observation_digest":"sha256:0e22ad4fa8b7d269edfd05a90736823b7fd4b6d79944a4a16700dd7ed7d1f52a","observation_id":"2d06fb11-d978-4def-ab53-7d7750db16ee","resolution":{"observed_at":"2026-08-15T19:09:44.454827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-06T22:47:52.271930Z","title":"D., Zhao, R., More, A., Hall, M., Khodamoradi, A., Deng, S., Choudhary, D., Cornea, M., Dellinger, E., Denolf, K., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20752","last_updated":"2025-06-25T18:25:08Z","snapshot_observed_at":"2026-08-18T20:54:07.774754Z","submitted_at":"2025-06-25T18:25:08Z","title":"Characterization and Mitigation of Training Instabilities in Microscaling Formats","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T22:47:52.271930Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2506.20752"},"observation_digest":"sha256:78224e587074cc624f5a0b948ce9b0ae4686eca3b7f7271040c0b546a79bb4d1","observation_id":"bc6af1a2-d133-4866-81ad-f447763dcaa4","resolution":{"observed_at":"2026-08-06T22:47:52.271930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-06T18:14:33.447407Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09010","last_updated":"2025-07-11T20:27:30Z","snapshot_observed_at":"2026-08-18T14:15:12.660551Z","submitted_at":"2025-07-11T20:27:30Z","title":"Hybrid Systolic Array Accelerator with Optimized Dataflow for Edge Large Language Model Inference","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T18:14:33.447407Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2507.09010"},"observation_digest":"sha256:f86533ddce13a1dd9069dc218ca79bba1cf3d40c2885691dc121d59c2b4597f1","observation_id":"260f865f-bf22-41ba-941c-9f2c8516265b","resolution":{"observed_at":"2026-08-06T18:14:33.447407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-06T15:24:17.313135Z","title":"D., Zhao, R., More, A., Hall, M., Khodamoradi, A., Deng, S., Choudhary, D., Cornea, M., Dellinger, E., Denolf, K., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.16099","last_updated":"2025-07-21T22:50:12Z","snapshot_observed_at":"2026-08-17T00:58:46.848934Z","submitted_at":"2025-07-21T22:50:12Z","title":"TorchAO: PyTorch-Native Training-to-Serving Model Optimization","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T15:24:17.313135Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2507.16099"},"observation_digest":"sha256:3b8e0aa25d04c196683e8ffce87af6c4f7f3033fcb0e71d37867df036215c4fb","observation_id":"b732f6c7-90c8-4a87-af9b-f6255f175922","resolution":{"observed_at":"2026-08-06T15:24:17.313135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2509.09505","last_updated":"2026-04-12T10:29:26Z","snapshot_observed_at":"2026-07-06T22:28:48.082201Z","submitted_at":"2025-09-11T14:49:50Z","title":"Combating the Memory Walls: Optimization Pathways for Long-Context Agentic LLM Inference","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-18T17:47:58.019030Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2509.09505"},"observation_digest":"sha256:7078d049ff40eff72bc9b90adc9b11d680c5338f2d51cd97aeebe371e6857158","observation_id":"8ecaf391-d4bf-46f3-b78c-4663723696ff","resolution":{"observed_at":"2026-05-18T17:51:42.214887Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2512.02010","last_updated":"2026-05-09T05:39:54Z","snapshot_observed_at":"2026-08-12T20:11:32.365290Z","submitted_at":"2025-12-01T18:59:45Z","title":"Four Over Six: More Accurate NVFP4 Quantization with Adaptive Block Scaling","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T02:23:01.845123Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2512.02010"},"observation_digest":"sha256:fad6874e962f36f5ffd7772138f6d1b07556a10960da709ec4e3fa434383521f","observation_id":"92e42020-4801-456c-9c6a-fbbfd4e89d65","resolution":{"observed_at":"2026-05-17T02:23:52.774424Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-03T16:35:53.163602Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.12930","last_updated":"2026-01-22T01:38:36Z","snapshot_observed_at":"2026-08-13T05:36:49.826841Z","submitted_at":"2025-12-15T02:29:08Z","title":"SeVeDo: A Heterogeneous Transformer Accelerator for Low-Bit Inference via Hierarchical Group Quantization and SVD-Guided Mixed Precision","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T16:35:53.163602Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2512.12930"},"observation_digest":"sha256:dff3537b79e01bbd02969fff38b951fab15b3d6ed394cd057299f840a4020fb8","observation_id":"5bacdbd1-1915-4f66-80a4-195e592fc989","resolution":{"observed_at":"2026-08-03T16:35:53.163602Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-03T11:10:52.658894Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.07475","last_updated":"2026-07-04T06:36:53Z","snapshot_observed_at":"2026-08-15T13:14:22.329443Z","submitted_at":"2026-01-12T12:27:22Z","title":"ARCQuant: Boosting NVFP4 Quantization with Augmented Residual Channels for LLMs","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-03T11:10:52.658894Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2601.07475"},"observation_digest":"sha256:336a233e363ec364446e94b35b5fc946677d2435dda84048e30b146a5f50eced","observation_id":"b922ee21-5e4e-48e3-84d8-c4dd2ff40647","resolution":{"observed_at":"2026-08-03T11:10:52.658894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-03T03:13:24.213416Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.08923","last_updated":"2026-07-07T20:21:24Z","snapshot_observed_at":"2026-08-19T21:36:05.531467Z","submitted_at":"2026-02-09T17:25:37Z","title":"DynamiQ: Accelerating Gradient Synchronization using Compressed Multi-hop All-reduce","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-03T03:13:24.213416Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2602.08923"},"observation_digest":"sha256:691a68f1bcfb2d4aa0ca8b936bc171eda52d76510aa2bfd0cd02959a071cee9c","observation_id":"26067140-9557-461d-97a0-fc26efc44524","resolution":{"observed_at":"2026-08-03T03:13:24.213416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2604.03950","last_updated":"2026-04-05T03:56:21Z","snapshot_observed_at":"2026-08-14T16:52:08.473786Z","submitted_at":"2026-04-05T03:56:21Z","title":"Diagonal-Tiled Mixed-Precision Attention for Efficient Low-Bit MXFP Inference","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T17:32:06.384120Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2604.03950"},"observation_digest":"sha256:e4076ce35aba330662e6208e1c4c10dfb261f0222d394bf3e9aca19774a1f934","observation_id":"9138fc19-8e25-4027-b643-e08a85025d27","resolution":{"observed_at":"2026-05-13T17:33:02.303676Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2604.04523","last_updated":"2026-04-06T08:35:41Z","snapshot_observed_at":"2026-08-14T08:00:11.832992Z","submitted_at":"2026-04-06T08:35:41Z","title":"LOCALUT: Harnessing Capacity-Computation Tradeoffs for LUT-Based Inference in DRAM-PIM","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-10T20:05:11.335715Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2604.04523"},"observation_digest":"sha256:aeb2bbee4f4cc3e9c1ecfd29ae090361c5fbf60b19190e1120737454e9c25e42","observation_id":"991abe71-dca1-4f68-bf04-e8755d7693d3","resolution":{"observed_at":"2026-05-10T22:15:49.669653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2604.08826","last_updated":"2026-04-09T23:50:56Z","snapshot_observed_at":"2026-08-12T17:13:49.040424Z","submitted_at":"2026-04-09T23:50:56Z","title":"HiFloat4 Format for Language Model Pre-training on Ascend NPUs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:49:58.735409Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2604.08826"},"observation_digest":"sha256:8991991eaf61e01258518a96f26822d24769cc617dcaebb76360a51ce1b17cf8","observation_id":"ff7dc2f2-3774-4385-9ac7-c5b2ebeaeddf","resolution":{"observed_at":"2026-05-11T08:05:59.824512Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2604.12782","last_updated":"2026-04-14T14:17:59Z","snapshot_observed_at":"2026-08-11T18:44:28.415280Z","submitted_at":"2026-04-14T14:17:59Z","title":"OSC: Hardware Efficient W4A4 Quantization via Outlier Separation in Channel Dimension","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:15:59.622461Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2604.12782"},"observation_digest":"sha256:5e1ce454b1c70171620886562f65275566fc15587f02efb682be5e1eddd52ca7","observation_id":"6fb18462-2aea-4ee2-b1c3-9dc2d7b98fd3","resolution":{"observed_at":"2026-05-11T09:05:59.520668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.02568","last_updated":"2026-05-04T13:19:29Z","snapshot_observed_at":"2026-08-11T08:33:37.167907Z","submitted_at":"2026-05-04T13:19:29Z","title":"StreamIndex: Memory-Bounded Compressed Sparse Attention via Streaming Top-k","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-08T18:44:42.456111Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.02568"},"observation_digest":"sha256:2fe64258dd9738d9151abffdfb369c92f78475653160979f5c5f2b479dc972d4","observation_id":"c48a6cc7-ca14-4642-9ecd-25e9b0f4f0e9","resolution":{"observed_at":"2026-05-09T06:15:37.680566Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.09825","last_updated":"2026-08-12T13:15:17Z","snapshot_observed_at":"2026-08-15T23:12:05.997262Z","submitted_at":"2026-05-11T00:04:07Z","title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-12T05:03:04.972344Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.09825"},"observation_digest":"sha256:4ef9d94cdf0f3a494c0457f109a0a36b692b19a61c64b5d06f5a7e583b7a164c","observation_id":"4e18bf1c-568d-4e90-9799-6564fc0f453e","resolution":{"observed_at":"2026-05-12T05:41:26.364189Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.09825","last_updated":"2026-08-12T13:15:17Z","snapshot_observed_at":"2026-08-15T23:12:05.997262Z","submitted_at":"2026-05-11T00:04:07Z","title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-14T21:04:57.021482Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.09825"},"observation_digest":"sha256:f3b9ae9cc550ed342b85999d9ab734c152244472e09bde6c6a263fe09e09c94b","observation_id":"2dd452da-9ee5-4db2-9ce8-19a2f9ab37c3","resolution":{"observed_at":"2026-05-14T21:19:28.727720Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.09825","last_updated":"2026-08-12T13:15:17Z","snapshot_observed_at":"2026-08-15T23:12:05.997262Z","submitted_at":"2026-05-11T00:04:07Z","title":"Pretraining large language models with MXFP4 on Native FP4 Hardware","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-15T05:15:58.477985Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.09825"},"observation_digest":"sha256:79842aee3563361fd9755d58e33d84538e134e69e2be4e0fd91c0ff44b1377b0","observation_id":"c9c9f698-f1df-4444-9864-de54501e14a4","resolution":{"observed_at":"2026-05-15T05:19:46.436060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.10886","last_updated":"2026-07-09T02:30:06Z","snapshot_observed_at":"2026-08-16T19:22:32.348351Z","submitted_at":"2026-05-11T17:32:29Z","title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-12T04:33:41.411292Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.10886"},"observation_digest":"sha256:f78f02905cca8c7d4a5283575bab603fd95df472005c4dc2c1eade7b01bfc84d","observation_id":"6f91f873-ce09-485d-9fa8-6aa85ab53d52","resolution":{"observed_at":"2026-05-12T06:06:28.153739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.10886","last_updated":"2026-07-09T02:30:06Z","snapshot_observed_at":"2026-08-16T19:22:32.348351Z","submitted_at":"2026-05-11T17:32:29Z","title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-15T04:55:01.973832Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.10886"},"observation_digest":"sha256:5bee8478843a9b2dab746e3822fa009a94c9f4fce418fd518bacc53fc87ce7d6","observation_id":"9b8803d6-86b6-4a38-96d5-abadc95c0592","resolution":{"observed_at":"2026-05-15T04:59:46.216734Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.11546","last_updated":"2026-05-12T05:20:43Z","snapshot_observed_at":"2026-08-17T16:25:52.311981Z","submitted_at":"2026-05-12T05:20:43Z","title":"The Entropy of Floating-Point Numbers","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T01:26:53.802738Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.11546"},"observation_digest":"sha256:8c3b20fffa4d62e1a1c5080a3079ad5ba41f19590356d2bccd5b02fc3ba3bf6c","observation_id":"c8b3b680-1f3f-4f4e-abf6-ef929b62cb3c","resolution":{"observed_at":"2026-05-13T01:27:01.603906Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.12245","last_updated":"2026-05-12T15:13:18Z","snapshot_observed_at":"2026-08-15T11:14:13.396806Z","submitted_at":"2026-05-12T15:13:18Z","title":"SOAR: Scale Optimization for Accurate Reconstruction in NVFP4 Quantization","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-13T05:54:01.264659Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.12245"},"observation_digest":"sha256:7b045aeed66ca1461c5a50b2c4b30e0003e4af8884ed9da563148171fe3c3132","observation_id":"d9b819fc-1efd-4f57-bca2-adf82abf8687","resolution":{"observed_at":"2026-05-13T06:02:23.831066Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.12327","last_updated":"2026-05-12T16:09:02Z","snapshot_observed_at":"2026-08-03T03:41:10.713228Z","submitted_at":"2026-05-12T16:09:02Z","title":"Grid Games: The Power of Multiple Grids for Quantizing Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T06:44:59.501345Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.12327"},"observation_digest":"sha256:4dfc2472fb64760355ce4af9d9eb00ac61b02fd56ce9bd9d953ae32873ae78dd","observation_id":"33871e95-793f-44b0-98f6-1c01b0f10fa3","resolution":{"observed_at":"2026-05-13T06:47:26.456839Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.13915","last_updated":"2026-05-13T09:49:56Z","snapshot_observed_at":"2026-08-14T16:52:11.220214Z","submitted_at":"2026-05-13T09:49:56Z","title":"Multi-Scale Dequant: Eliminating Dequantization Bottleneck via Activation Decomposition for Efficient LLM Inference","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:18.944617Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.13915"},"observation_digest":"sha256:816c81a288707b383f1ec0f0e20fc5cfb97b90a7dec5a44829813dc695987bcc","observation_id":"0d46cba1-eb42-4a06-a9c9-599a942712af","resolution":{"observed_at":"2026-05-15T02:58:34.206283Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.14929","last_updated":"2026-05-14T15:03:58Z","snapshot_observed_at":"2026-08-14T16:51:24.575848Z","submitted_at":"2026-05-14T15:03:58Z","title":"A Hardware-Aware, Per-Layer Methodology for Post-Training Quantization of Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-30T21:03:05.361805Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.14929"},"observation_digest":"sha256:305eb3d5be46b33d376a1e093df99216a6789c6217546c8f4153ac23bc778785","observation_id":"d601843e-fc0b-4f28-aa46-e865bb0f7d35","resolution":{"observed_at":"2026-06-30T21:05:03.956703Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.18739","last_updated":"2026-05-19T17:46:12Z","snapshot_observed_at":"2026-07-06T23:29:33.702647Z","submitted_at":"2026-05-18T17:57:03Z","title":"LongLive-2.0: An NVFP4 Parallel Infrastructure for Long Video Generation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-20T11:02:39.465293Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.18739"},"observation_digest":"sha256:8f1749d75181f8f9150cc24ac37cabbd737e9488520da85e5eea97b12659b9c6","observation_id":"17ee07c3-6a31-4afa-8d91-289e28cc800f","resolution":{"observed_at":"2026-05-20T11:03:13.294669Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.19195","last_updated":"2026-05-18T23:51:02Z","snapshot_observed_at":"2026-08-15T04:12:59.638392Z","submitted_at":"2026-05-18T23:51:02Z","title":"The Thermodynamic Costs of Simple Linear Regression","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-20T07:03:21.987679Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.19195"},"observation_digest":"sha256:02a8551de33ba8d495ef1008d26f011f237a54d800f49b8a816c86e14570f1b0","observation_id":"69db121e-6ddc-4936-a720-0d3b37141821","resolution":{"observed_at":"2026-05-20T07:03:23.258985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.23081","last_updated":"2026-05-21T22:28:27Z","snapshot_observed_at":"2026-08-13T02:49:10.130417Z","submitted_at":"2026-05-21T22:28:27Z","title":"ThriftAttention: Selective Mixed Precision for Long-Context FP4 Attention","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T05:28:40.888215Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.23081"},"observation_digest":"sha256:ae03e57141edcbf2b6c27f48337155650f4827c1e9e28c2648ba623c2beeafb0","observation_id":"aad4aede-8983-4ecf-87b1-85f69f9805d8","resolution":{"observed_at":"2026-05-25T05:30:22.804401Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.23226","last_updated":"2026-05-22T04:37:51Z","snapshot_observed_at":"2026-08-14T16:52:13.614659Z","submitted_at":"2026-05-22T04:37:51Z","title":"MASQ: Accelerating Masked Diffusion via Stage-Wise Multi-Precision Quantization","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-25T02:59:25.037162Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.23226"},"observation_digest":"sha256:cc4a26b5368f3379e9b3f9a77e58ecbc72b4cf42421459157ddc0897fee1ed9f","observation_id":"135f833e-3ba9-43e5-af4f-4ea77dfc875e","resolution":{"observed_at":"2026-05-25T03:00:15.853710Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.24391","last_updated":"2026-06-01T18:51:49Z","snapshot_observed_at":"2026-08-16T21:51:31.848849Z","submitted_at":"2026-05-23T04:21:57Z","title":"MX-SAFE: Versatile Inference- and Training-Proof Microscaling Format with On-the-Fly Exponent and Mantissa Bit Allocation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T12:42:57.909391Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.24391"},"observation_digest":"sha256:1ae55eb70f6eeccdde1376ca3f9cc4bbaee1249c4204487db29b6d37e03100d9","observation_id":"b373f641-574f-42ba-917d-93f9d7edce77","resolution":{"observed_at":"2026-06-30T12:44:39.221357Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2605.26558","last_updated":"2026-05-26T05:12:44Z","snapshot_observed_at":"2026-08-12T18:15:32.484112Z","submitted_at":"2026-05-26T05:12:44Z","title":"Cassandra: Enabling Reasoning LLMs at Edge via Self-Speculative Decoding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-01T16:19:54.910430Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2605.26558"},"observation_digest":"sha256:59e9091d7163c75c89186f3336a501b99a4e22ce7b9f7e8838509346c656a725","observation_id":"bb2595bb-1b2b-4090-9cd2-22061cbbe22e","resolution":{"observed_at":"2026-07-01T16:25:49.798946Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.02333","last_updated":"2026-06-01T14:44:05Z","snapshot_observed_at":"2026-08-03T13:27:52.287848Z","submitted_at":"2026-06-01T14:44:05Z","title":"O-POPE: High-Frequency Pipelined Outer Product based GEMM acceleration with minimal buffering overhead","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T12:11:38.789928Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.02333"},"observation_digest":"sha256:7d25ac38b4290318b7824bd79580e0689b772bdffa5cc275f8b39c5935340708","observation_id":"05e0a655-bb1e-40b1-ba1c-b15cc2f4fa1c","resolution":{"observed_at":"2026-07-02T01:16:25.114709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.04115","last_updated":"2026-07-14T14:04:32Z","snapshot_observed_at":"2026-08-02T05:34:23.902031Z","submitted_at":"2026-06-02T18:23:20Z","title":"dMX: Differentiable Mixed-Precision Assignment for Low-Precision Floating-Point Formats","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T11:20:06.292977Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.04115"},"observation_digest":"sha256:95564e4cdb2bf237eb9e3130f230e188702af80d606e4b2a8c24ba5867887df7","observation_id":"a1c44425-878a-4bf9-ad33-a0bc5552a848","resolution":{"observed_at":"2026-07-02T02:06:26.120872Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-15T10:56:54.155632Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.04115","last_updated":"2026-07-14T14:04:32Z","snapshot_observed_at":"2026-08-02T05:34:23.902031Z","submitted_at":"2026-06-02T18:23:20Z","title":"dMX: Differentiable Mixed-Precision Assignment for Low-Precision Floating-Point Formats","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-15T10:56:54.155632Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.04115"},"observation_digest":"sha256:8c33c9b071c399fe3bb13fb29cd99936be9c8980e19b37715f018f6e80d55469","observation_id":"481d94f3-5b30-4a98-b9f7-de44a7593944","resolution":{"observed_at":"2026-07-15T10:56:54.155632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.05017","last_updated":"2026-06-22T12:03:17Z","snapshot_observed_at":"2026-08-12T13:41:31.675123Z","submitted_at":"2026-06-03T15:41:16Z","title":"GoldenFloat: A Phi-Derived Static-Split Floating-Point Family from GF4 to GF1024 with a Lucas-Exact Integer Identity","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T03:47:27.000639Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.05017"},"observation_digest":"sha256:6ce8ca820b547059c865d484709fd1e4ba1e1b493499f412e9899e691617fc03","observation_id":"e902f905-60a7-459b-8bd5-03b9000ab0a7","resolution":{"observed_at":"2026-07-02T11:26:54.592249Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.05206","last_updated":"2026-05-23T19:19:32Z","snapshot_observed_at":"2026-08-12T18:13:45.963723Z","submitted_at":"2026-05-23T19:19:32Z","title":"Ontology-constrained multi-LLM scoring of hypothesis support in the predictive processing literature","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-30T11:49:43.332490Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.05206"},"observation_digest":"sha256:2aadea667425ac45d6f20470673cf2a6b3743d0881c64e05493718ef13b6383c","observation_id":"0377c884-1de7-4aca-9b77-524deeb31b7d","resolution":{"observed_at":"2026-06-30T11:54:37.877107Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.09686","last_updated":"2026-06-22T12:28:45Z","snapshot_observed_at":"2026-08-15T23:35:26.268440Z","submitted_at":"2026-06-08T16:04:15Z","title":"An 83-Format Numeric Catalog with Bit-Exact Conformance Vectors: A Vendor-Neutral Reference for FP8, BF16, MXFP4, and Microscaling Formats","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T14:33:42.876697Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.09686"},"observation_digest":"sha256:9ed386e5b2efbf4b77da344022146b6125b12cdb9cb52b1ec66ee45c882814c2","observation_id":"3be37054-5140-45bf-b02a-c094514b5651","resolution":{"observed_at":"2026-07-03T03:47:35.977373Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.13233","last_updated":"2026-06-11T11:47:40Z","snapshot_observed_at":"2026-08-14T12:06:54.132201Z","submitted_at":"2026-06-11T11:47:40Z","title":"ReSET: Accurate Latency-Critical NVFP4 Reasoning via Step-Aware Temperature Scaling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T07:40:01.154810Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.13233"},"observation_digest":"sha256:9dd80b49f36e1681edd2bb9313c03a8264dffcf28db0e3569e42fa46cd500ac3","observation_id":"1afda128-a2e0-45ec-9c83-f49e7ca59e58","resolution":{"observed_at":"2026-07-03T13:38:19.595324Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.20381","last_updated":"2026-06-18T15:40:51Z","snapshot_observed_at":"2026-08-13T12:46:13.058236Z","submitted_at":"2026-06-18T15:40:51Z","title":"Rethinking Shrinkage Bias in LLM FP4 Pretraining: Geometric Origin, Systemic Impact, and UFP4 Recipe","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-06-26T16:59:57.067069Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.20381"},"observation_digest":"sha256:5a9e76f852a01ab4ac44f2361a71fd3e5c24a27951995a527e353a9db5439fac","observation_id":"10c04bb2-3aca-450f-be0b-9ac900872458","resolution":{"observed_at":"2026-07-04T04:29:34.888762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.23406","last_updated":"2026-06-22T14:30:47Z","snapshot_observed_at":"2026-08-07T08:16:17.853171Z","submitted_at":"2026-06-22T14:30:47Z","title":"HyperQuant: A Rate-Distortion-Optimal Quantization Pipeline for Large Language and Diffusion Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-26T08:46:01.500880Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.23406"},"observation_digest":"sha256:fd195330ebacb3b8b28b683963a339b4266cad356e0867fae2907d0787f0fe2a","observation_id":"a1689784-7c2d-4ab1-a94d-10375afd1e29","resolution":{"observed_at":"2026-07-04T10:29:45.538537Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":"2310.10537","doi":"10.48550/arxiv.2310.10537","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537","venue":"arXiv (Cornell University)","work_id":"fead0f60-0895-4dc8-9fec-b579bf79288e","year":2023},"citing_paper":{"arxiv_id":"2606.26587","last_updated":"2026-06-25T04:19:04Z","snapshot_observed_at":"2026-08-17T15:55:33.333568Z","submitted_at":"2026-06-25T04:19:04Z","title":"SharQ: Bridging Activation Sparsity and FP4 Quantization for LLM Inference","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-26T05:41:39.052865Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2606.26587"},"observation_digest":"sha256:3fb35dd2b5408cea1f8d3c05f05b70d000f533cb06fd01ae8530e67869736408","observation_id":"a5cc00b9-0efd-4b55-aa55-d5d2d391f9cd","resolution":{"observed_at":"2026-07-04T12:59:52.348306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-14T20:38:10.394213+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-11T19:17:59.044982Z","title":"Microscaling data formats for deep learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.04422","last_updated":"2026-08-07T18:56:54Z","snapshot_observed_at":"2026-08-13T23:16:52.560110Z","submitted_at":"2026-07-05T17:31:36Z","title":"Full-Stack FP4: Stable LLM Pretraining with Quantized Projections, Optimizers, and Attention","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:59.044982Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.04422"},"observation_digest":"sha256:5860595e49695a21fb1094d851b9395219a8a77c26454f9281e1153ca2fcc22b","observation_id":"024a417c-01ec-4800-92ba-301d29e9da18","resolution":{"observed_at":"2026-07-11T19:17:59.044982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-14T15:56:34.283388Z","title":"Microscaling Data Formats for Deep Learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09775","last_updated":"2026-07-07T20:26:25Z","snapshot_observed_at":"2026-08-18T14:12:46.972982Z","submitted_at":"2026-07-07T20:26:25Z","title":"WINT: A Novel Weighted Integer Representation with Improved Error Characteristics","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-14T15:56:34.283388Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.09775"},"observation_digest":"sha256:f7b66d1024d33f3f08af7b434dd2e7c9697e32fbc71a5442669c2b55188aebb5","observation_id":"0988e4e9-1dd2-4acc-90a4-5efb10eec4d8","resolution":{"observed_at":"2026-07-14T15:56:34.283388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-02T03:25:29.316109Z","title":"Microscaling Data Formats for Deep Learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13898","last_updated":"2026-07-15T14:40:59Z","snapshot_observed_at":"2026-08-18T14:20:47.196766Z","submitted_at":"2026-07-15T14:40:59Z","title":"Jack of All Scales: A Versatile FPGA Tensor Block for MXFP Precisions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T03:25:29.316109Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.13898"},"observation_digest":"sha256:7008b218fd4f563b04417b924bacbb5d3a5d55431bd6ec59603e9f190d1c6a62","observation_id":"345071d8-d100-4f97-9f30-ebfbf078e0d6","resolution":{"observed_at":"2026-08-02T03:25:29.316109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-01T17:12:27.562005Z","title":"arXiv preprint arXiv:2310.10537 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17733","last_updated":"2026-07-20T09:23:10Z","snapshot_observed_at":"2026-08-20T01:37:42.473116Z","submitted_at":"2026-07-20T09:23:10Z","title":"MXSens: Sensitivity-Aware Mixed-Precision Quantization for Efficient LLM Inference","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-01T17:12:27.562005Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.17733"},"observation_digest":"sha256:39ed88b88f879bca2548cd2435726e745a00d7834cd3bc8ed74ac8bb0c008cb3","observation_id":"be13ca87-9323-40a3-bad4-990ba4de8185","resolution":{"observed_at":"2026-08-01T17:12:27.562005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-01T17:12:28.283144Z","title":"Microscaling Data Formats for Deep Learning , journal =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17733","last_updated":"2026-07-20T09:23:10Z","snapshot_observed_at":"2026-08-20T01:37:42.473116Z","submitted_at":"2026-07-20T09:23:10Z","title":"MXSens: Sensitivity-Aware Mixed-Precision Quantization for Efficient LLM Inference","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-01T17:12:28.283144Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.17733"},"observation_digest":"sha256:8e95b58d456d47f54dbc5a08ce9eefb1c682b79de0a199f8396d9b9d26756176","observation_id":"d150366b-146d-4864-8500-6d7060c83dfc","resolution":{"observed_at":"2026-08-01T17:12:28.283144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-02T08:11:51.850576Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.20518","last_updated":"2026-07-08T02:22:22Z","snapshot_observed_at":"2026-08-14T17:25:19.834750Z","submitted_at":"2026-07-08T02:22:22Z","title":"CANN Bench: Benchmarking Agent Generated Kernels against Real NPU and Algorithmic Limits","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T08:11:51.850576Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.20518"},"observation_digest":"sha256:58b6ce6f4621f3b9c75138a53c6540d015e72c9db49a045143e4956371cbc6b5","observation_id":"16918522-e25b-46b1-aa71-5d4ea41f551a","resolution":{"observed_at":"2026-08-02T08:11:51.850576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-31T16:32:52.030874Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24377","last_updated":"2026-07-27T12:53:22Z","snapshot_observed_at":"2026-08-13T16:58:15.773596Z","submitted_at":"2026-07-27T12:53:22Z","title":"MXAttention: Data-Free Optimal Scaling and Pre-Normalization Quantization for MXFP4 Attention","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-31T16:32:52.030874Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.24377"},"observation_digest":"sha256:952067d3819e77b6c36bc0d8d13ca74268ff392e9ce55d1f9ee5c96d2370f415","observation_id":"e42587c1-dcac-4350-8f5e-00f64a879e1f","resolution":{"observed_at":"2026-07-31T16:32:52.030874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-07-31T05:04:00.375801Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24953","last_updated":"2026-07-27T18:03:34Z","snapshot_observed_at":"2026-08-17T22:46:35.095500Z","submitted_at":"2026-07-27T18:03:34Z","title":"Stable FP4 Training via Transposition-Invariant Block Quantization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-31T05:04:00.375801Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.24953"},"observation_digest":"sha256:01a3178e9cb0ddbb1cb48288ac62247515d237cc42b75ee85bfd98e6cc3209c7","observation_id":"aca84154-2b5c-41cb-816f-ca068542a253","resolution":{"observed_at":"2026-07-31T05:04:00.375801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-01T03:16:46.884728Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.27694","last_updated":"2026-07-30T05:26:20Z","snapshot_observed_at":"2026-08-19T20:06:11.260907Z","submitted_at":"2026-07-30T05:26:20Z","title":"GyRot: Leveraging Hidden Synergy between Rotation and Fine-grained Group Quantization for Low-bit LLM Inference","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-01T03:16:46.884728Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.27694"},"observation_digest":"sha256:f43ffcc916831739e6960d058022b8b964e327fa670249d94371312f21d5a16e","observation_id":"2d02397a-cb4b-4f3f-a300-63f457fec6d0","resolution":{"observed_at":"2026-08-01T03:16:46.884728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-01T03:01:45.331505Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27704","last_updated":"2026-07-30T05:39:58Z","snapshot_observed_at":"2026-08-14T14:39:11.713519Z","submitted_at":"2026-07-30T05:39:58Z","title":"LightRot: A Light-Weighted Rotation Scheme and Architecture for Accurate Low-Bit Large Language Model Inference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T03:01:45.331505Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.27704"},"observation_digest":"sha256:bb256cd8d0317a7ecd352b492731478ffc70c572a1dac183c14bf301054c9eda","observation_id":"2399894d-50e4-4d9a-9916-01847cc72137","resolution":{"observed_at":"2026-08-01T03:01:45.331505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-03T07:51:22.144778Z","title":"arXiv preprint arXiv:2310.10537 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29397","last_updated":"2026-08-04T16:14:13Z","snapshot_observed_at":"2026-08-13T00:21:03.167709Z","submitted_at":"2026-07-31T13:15:11Z","title":"Studying quantization trade-offs for efficient inference deployment in machine translation","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-03T07:51:22.144778Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.29397"},"observation_digest":"sha256:7b2d346876b57f9584eeb5b9e1d61afb954bb11128147817079ba92fc182ad54","observation_id":"07becc51-2ceb-46cf-a8fa-f4e98530c00b","resolution":{"observed_at":"2026-08-03T07:51:22.144778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T04:25:22.011849Z","title":"arXiv preprint arXiv:2310.10537 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29397","last_updated":"2026-08-04T16:14:13Z","snapshot_observed_at":"2026-08-13T00:21:03.167709Z","submitted_at":"2026-07-31T13:15:11Z","title":"Studying quantization trade-offs for efficient inference deployment in machine translation","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T04:25:22.011849Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2607.29397"},"observation_digest":"sha256:75d2296fb47f49c8c555528ee36641b51ffec586560a88cd8cf96b71fb43c285","observation_id":"a406126b-0fa0-4da3-ba75-2c0fab223ca1","resolution":{"observed_at":"2026-08-05T04:25:22.011849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-04T19:49:24.965978Z","title":"arXiv preprint arXiv:2310.10537 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.01847","last_updated":"2026-08-03T07:54:56Z","snapshot_observed_at":"2026-08-19T04:12:33.131408Z","submitted_at":"2026-08-03T07:54:56Z","title":"FOCUS: FP4 Optimization via Coupled-Relaxation and Dual-Granularity Scaling","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-04T19:49:24.965978Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.01847"},"observation_digest":"sha256:432ced015c9bd6c936985132eab7e126b2ac705ea4744176523cd61a0dcee463","observation_id":"af45e318-48cd-4b14-b734-4317077b8f5e","resolution":{"observed_at":"2026-08-04T19:49:24.965978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-04T15:20:16.175942Z","title":"Microscaling data formats for deep learning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02091","last_updated":"2026-08-03T11:50:06Z","snapshot_observed_at":"2026-08-17T04:16:46.701964Z","submitted_at":"2026-08-03T11:50:06Z","title":"One QK Channel, Many Sources: Guarding Low-Precision Attention Collapse","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T15:20:16.175942Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.02091"},"observation_digest":"sha256:597cb54f6314eb18e670e69cb122d81ca85b0a22a080ec7d938f534d10b24b47","observation_id":"0feeb5e7-3265-4323-a76d-44f7884cf796","resolution":{"observed_at":"2026-08-04T15:20:16.175942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T13:48:36.545613Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.03741","last_updated":"2026-08-04T14:35:25Z","snapshot_observed_at":"2026-08-19T10:59:34.130592Z","submitted_at":"2026-08-04T14:35:25Z","title":"When Does Disaggregation Pay? Simulating Prefill--Decode--Attention--FFN Specialization for Agentic LLM Inference","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:36.545613Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.03741"},"observation_digest":"sha256:4934c486e49cb05aa9b1c91c60c7b932b58a731054b11d6babf86b995181eff0","observation_id":"4833d5a2-b500-44c4-a4c2-19fff570c7ee","resolution":{"observed_at":"2026-08-05T13:48:36.545613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-05T10:54:30.865642Z","title":"Microscaling data formats for deep learning,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.03867","last_updated":"2026-08-04T16:08:20Z","snapshot_observed_at":"2026-08-18T06:24:20.935343Z","submitted_at":"2026-08-04T16:08:20Z","title":"Heterogeneity-Aware Microscaling for Efficient Low-Bit LLM Inference","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T10:54:30.865642Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.03867"},"observation_digest":"sha256:3e7c6b7e50acc007555d33c257ec58e9c3f530d2ce8fc1a7ae7e3297b80e1e41","observation_id":"f8149f9c-a0c1-4b1e-84be-2c3e9c5b26d4","resolution":{"observed_at":"2026-08-05T10:54:30.865642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-11T23:12:10.143929Z","title":"Microscaling data formats for deep learning.arXiv preprint arXiv:2310.10537, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09119","last_updated":"2026-08-10T04:53:05Z","snapshot_observed_at":"2026-08-19T16:33:29.003229Z","submitted_at":"2026-08-10T04:53:05Z","title":"Motif 3: Technical Report","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-11T23:12:10.143929Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.09119"},"observation_digest":"sha256:4727cc8270470c0bc95e97c19e8cbf6dbb5a3e73da196dee4ba151769f062286","observation_id":"88f16c11-468e-465d-97e0-d854675b5a6f","resolution":{"observed_at":"2026-08-11T23:12:10.143929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-12T00:51:15.605710Z","title":"URL https://arxiv.org/ abs/2310.10537","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10010","last_updated":"2026-08-12T03:58:39Z","snapshot_observed_at":"2026-08-17T09:38:05.904832Z","submitted_at":"2026-08-08T01:55:46Z","title":"CurveFP: Co-Designing Numerical Representation and Product Arithmetic for Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T00:51:15.605710Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.10010"},"observation_digest":"sha256:e674f4cb805c041210f23256d7d5a1fe73e5ed5837e4d4ffb7fd531d2e763079","observation_id":"73f6dc94-4c05-4d38-87bb-2e905ce944fd","resolution":{"observed_at":"2026-08-12T00:51:15.605710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.10537","snapshot_observed_at":"2026-08-14T04:46:37.255337Z","title":"URL https://arxiv.org/ abs/2310.10537","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10010","last_updated":"2026-08-12T03:58:39Z","snapshot_observed_at":"2026-08-17T09:38:05.904832Z","submitted_at":"2026-08-08T01:55:46Z","title":"CurveFP: Co-Designing Numerical Representation and Product Arithmetic for Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-14T04:46:37.255337Z"},"links":{"cited_paper":"/paper/2310.10537","citing_paper":"/paper/2608.10010"},"observation_digest":"sha256:d38a3045651590a3d90ff6c797f5aa333d9e34166356821381a1eed6cfc0f6b8","observation_id":"d743f1df-8ce5-427e-802b-41fe267b5fcc","resolution":{"observed_at":"2026-08-14T04:46:37.255337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.10537/citation-record","integrity":"/paper/2310.10537/integrity","json":"/paper/2310.10537/citation-record.json","paper":"/paper/2310.10537"},"outbound":[],"paper":{"arxiv_id":"2310.10537","last_updated":"2023-10-19T16:38:33Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-19T14:12:52.683965Z","submitted_at":"2023-10-16T16:07:41Z","title":"Microscaling Data Formats for Deep Learning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 78 inbound Pith citation observations for arXiv:2310.10537."}