{"as_of":"2026-08-21T03:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e3e9f05a8f8221977b312b9aa42ba8c5578eee01e9afcabf8e1860d5ee54e694","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":38,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T17:55:13.736803Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:29:44.312964Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2203.04153","last_updated":"2026-04-25T00:38:05Z","snapshot_observed_at":"2026-08-12T21:16:31.903561Z","submitted_at":"2022-03-08T15:30:32Z","title":"Easy Ensemble: Simple Deep Ensemble Learning for Sensor-Based Human Activity Recognition","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-24T11:20:49.217671Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2203.04153"},"observation_digest":"sha256:787ca6662cec9a5c0638ff1b39106b62bcb5c37856fd03069079d57b482c3ad4","observation_id":"fa908253-a6f3-49aa-9a8a-f735aef5c6eb","resolution":{"observed_at":"2026-05-24T11:24:24.068765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2305.07759","last_updated":"2023-05-24T23:30:43Z","snapshot_observed_at":"2026-08-20T14:47:29.320021Z","submitted_at":"2023-05-12T20:56:48Z","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T07:36:55.087443Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2305.07759"},"observation_digest":"sha256:5159c2862931a4fa05c79c84c365777d41bc6053d06eac670e23cca8ffa52572","observation_id":"9d7127de-539c-4440-b771-13a68a10e26f","resolution":{"observed_at":"2026-05-25T07:36:55.226039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2401.01335","last_updated":"2024-06-14T21:17:17Z","snapshot_observed_at":"2026-08-19T08:48:56.371619Z","submitted_at":"2024-01-02T18:53:13Z","title":"Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-14T23:00:20.720030Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2401.01335"},"observation_digest":"sha256:d8e8884a00b9f901b65150a94795e9d668242160ed705b2c33f57e35e182e0ff","observation_id":"e26a1e47-2346-4b40-8a38-178e9a2f7570","resolution":{"observed_at":"2026-05-14T23:00:20.988097Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-12T00:11:14.614276Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning.arXiv:2012.09816,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2412.01951","last_updated":"2024-12-04T14:20:21Z","snapshot_observed_at":"2026-08-18T01:49:25.380130Z","submitted_at":"2024-12-02T20:24:17Z","title":"Self-Improvement in Language Models: The Sharpening Mechanism","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-12T00:11:14.614276Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2412.01951"},"observation_digest":"sha256:ee91ca21e690246ab2c774773f814d3b95351409f4c7bada385b5db01e22fc5a","observation_id":"b33b96a4-3cf4-41af-84b2-d8c087597fa8","resolution":{"observed_at":"2026-08-12T00:11:14.614276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-11T13:40:35.424919Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.12968","last_updated":"2025-01-07T14:45:04Z","snapshot_observed_at":"2026-08-14T12:40:34.288551Z","submitted_at":"2024-12-17T14:53:38Z","title":"On Local Overfitting and Forgetting in Deep Neural Networks","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T13:40:35.424919Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2412.12968"},"observation_digest":"sha256:fc98fdb00fc107cf03877f3d20b93d8452e28a294a4881cc9716c050c5f752be","observation_id":"2b69744c-abff-4a85-a9ab-52589e056b54","resolution":{"observed_at":"2026-08-11T13:40:35.424919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-11T22:42:01.534949Z","title":"Towards understanding ensembl e, knowledge distillation and self-distillation in deep learning,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2412.15224","last_updated":"2024-12-04T11:31:23Z","snapshot_observed_at":"2026-08-17T07:04:25.640977Z","submitted_at":"2024-12-04T11:31:23Z","title":"Multi-Branch Mutual-Distillation Transformer for EEG-Based Seizure Subtype Classification","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T22:42:01.534949Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2412.15224"},"observation_digest":"sha256:f6f6e367d50b6f93de290068fb14c38589e34fe6449b31b79c5128fc0d967d92","observation_id":"0db7fd84-b5f2-4bac-9003-2cc1f562b098","resolution":{"observed_at":"2026-08-11T22:42:01.534949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-10T13:57:21.164110Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2501.15925","last_updated":"2025-05-28T03:23:09Z","snapshot_observed_at":"2026-08-14T21:38:15.888564Z","submitted_at":"2025-01-27T10:22:38Z","title":"Efficient Logit-based Knowledge Distillation of Deep Spiking Neural Networks for Full-Range Timestep Deployment","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T13:57:21.164110Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2501.15925"},"observation_digest":"sha256:4957b1acf93f64b4695bf33563d3d830b067beb6b298661448f97251b8d42ec1","observation_id":"78188984-e501-4903-ae45-27ab1ba7616b","resolution":{"observed_at":"2026-08-10T13:57:21.164110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-10T13:34:03.502948Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2501.16329","last_updated":"2025-01-27T18:59:55Z","snapshot_observed_at":"2026-08-18T02:21:30.093341Z","submitted_at":"2025-01-27T18:59:55Z","title":"sDREAMER: Self-distilled Mixture-of-Modality-Experts Transformer for Automatic Sleep Staging","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T13:34:03.502948Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2501.16329"},"observation_digest":"sha256:66c6f9d0cafd16cfbfe7cb848e04e576880de915d96e8673613ade542644090e","observation_id":"01614683-ebbf-4344-8e6f-8b0787e9de1b","resolution":{"observed_at":"2026-08-10T13:34:03.502948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-09T18:24:40.048724Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.00620","last_updated":"2025-06-17T22:06:53Z","snapshot_observed_at":"2026-08-14T10:13:58.922061Z","submitted_at":"2025-02-02T01:11:51Z","title":"Representations Shape Weak-to-Strong Generalization: Theoretical Insights and Empirical Predictions","version":4},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T18:24:40.048724Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2502.00620"},"observation_digest":"sha256:02b53957a62b682edbb4047281dc93b7ca0bfcce0f61ec4082a9940900f905ca","observation_id":"81093135-1811-4529-b77c-43b1a131e698","resolution":{"observed_at":"2026-08-09T18:24:40.048724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-08T21:50:24.493688Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.04725","last_updated":"2025-02-07T07:49:37Z","snapshot_observed_at":"2026-08-14T17:03:42.125720Z","submitted_at":"2025-02-07T07:49:37Z","title":"Can Diffusion Models Learn Hidden Inter-Feature Rules Behind Images?","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-08T21:50:24.493688Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2502.04725"},"observation_digest":"sha256:db87243f8e9507cd4b642ba6c44e1fb8ae86cb119ce4c814c92f7b7ed76220dd","observation_id":"ff5ccdee-d2c8-4cda-a074-b70e877f76b2","resolution":{"observed_at":"2026-08-08T21:50:24.493688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-08T16:31:36.475961Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.06192","last_updated":"2025-05-19T14:51:05Z","snapshot_observed_at":"2026-08-16T03:17:17.427325Z","submitted_at":"2025-02-10T06:48:04Z","title":"Right Time to Learn:Promoting Generalization via Bio-inspired Spacing Effect in Knowledge Distillation","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-08T16:31:36.475961Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2502.06192"},"observation_digest":"sha256:f999e4b5cdabe10cc0b3eec259bca97d8804f5ecb8f226d9ebbb627115375ff1","observation_id":"028bb0b8-5f8f-4062-8df1-89ad6f74f7bb","resolution":{"observed_at":"2026-08-08T16:31:36.475961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-08T21:40:12.790902Z","title":"CoRRabs/2012.09816 (2020), https://arxiv","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06849","last_updated":"2025-02-07T08:45:38Z","snapshot_observed_at":"2026-08-15T23:40:27.267103Z","submitted_at":"2025-02-07T08:45:38Z","title":"Model Fusion via Neuron Transplantation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T21:40:12.790902Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2502.06849"},"observation_digest":"sha256:0a3a7bf19efceae27b16076f4a04f89bf5e5e6944758c13da013365b6b32a6cc","observation_id":"2b59b83d-946c-452b-a971-f18dc250df86","resolution":{"observed_at":"2026-08-08T21:40:12.790902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T11:39:51.726608Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.01901","last_updated":"2025-06-02T17:23:16Z","snapshot_observed_at":"2026-08-20T03:19:37.397089Z","submitted_at":"2025-06-02T17:23:16Z","title":"Understanding Overadaptation in Supervised Fine-Tuning: The Role of Ensemble Methods","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T11:39:51.726608Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.01901"},"observation_digest":"sha256:08e175863c4d00a62af52fddacae8ddc8895ddcefbd709583f5a7d65ef68933c","observation_id":"fbe9a7a8-7549-4797-90cc-c253e3ea3f93","resolution":{"observed_at":"2026-08-07T11:39:51.726608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T11:26:13.168645Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.02557","last_updated":"2025-06-03T07:44:43Z","snapshot_observed_at":"2026-08-12T03:14:23.179364Z","submitted_at":"2025-06-03T07:44:43Z","title":"Kernel-based Unsupervised Embedding Alignment for Enhanced Visual Representation in Vision-language Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T11:26:13.168645Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.02557"},"observation_digest":"sha256:c1e1a92f5d7d36f708e427058e2aa77a14136955481a78e7105de055e9d612ec","observation_id":"9ac6405e-c393-4fe7-be5e-44e9302a72b6","resolution":{"observed_at":"2026-08-07T11:26:13.168645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T10:31:20.328935Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05447","last_updated":"2025-07-14T23:29:38Z","snapshot_observed_at":"2026-08-18T01:42:19.840722Z","submitted_at":"2025-06-05T15:18:35Z","title":"Training Dynamics Underlying Language Model Scaling Laws: Loss Deceleration and Zero-Sum Learning","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T10:31:20.328935Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.05447"},"observation_digest":"sha256:7216c3ca61b93150e1105abcce5fc03fbeb80a5deb313713895069c38ee7f220","observation_id":"187ee80c-5b8b-4d2e-bde7-363c5159e6eb","resolution":{"observed_at":"2026-08-07T10:31:20.328935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T04:17:21.874305Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.10956","last_updated":"2025-06-12T17:55:48Z","snapshot_observed_at":"2026-08-17T11:14:53.436730Z","submitted_at":"2025-06-12T17:55:48Z","title":"Distillation of atomistic foundation models across architectures and chemical domains","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T04:17:21.874305Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.10956"},"observation_digest":"sha256:a4ee37ef59188572d5459eb999975f8fa39833f6450b00b04724b86e7824610d","observation_id":"fa11f9c9-8ab7-4336-8494-efbd30b1a47b","resolution":{"observed_at":"2026-08-07T04:17:21.874305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-07T01:03:46.623707Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2506.12226","last_updated":"2025-06-13T21:03:49Z","snapshot_observed_at":"2026-08-13T16:48:11.025073Z","submitted_at":"2025-06-13T21:03:49Z","title":"Learning Causality for Modern Machine Learning","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T01:03:46.623707Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.12226"},"observation_digest":"sha256:15e41c4f790fadbc013932acf3c41dadeae6d3a09dd265224cfee73286760f53","observation_id":"b0c73fea-4d4e-4224-a0ba-7902fca8efe4","resolution":{"observed_at":"2026-08-07T01:03:46.623707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-06T21:31:35.399144Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.23990","last_updated":"2025-06-30T15:55:10Z","snapshot_observed_at":"2026-08-14T06:44:03.414206Z","submitted_at":"2025-06-30T15:55:10Z","title":"Machine Understanding of Scientific Language","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:31:35.399144Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2506.23990"},"observation_digest":"sha256:5aa59dc27c3ff7879fd3064fac27eff9897870867cd16e21b9a58647151c2548","observation_id":"14de3fd5-95fe-467c-8942-efefb6e14fdb","resolution":{"observed_at":"2026-08-06T21:31:35.399144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-06T18:19:59.213224Z","title":"Towards understanding ensem- ble, knowledge distillation and self-distillation in deep learning,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.08686","last_updated":"2025-07-11T15:37:24Z","snapshot_observed_at":"2026-08-10T14:00:04.216873Z","submitted_at":"2025-07-11T15:37:24Z","title":"Forget Me Not: Fighting Local Overfitting with Knowledge Fusion and Distillation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T18:19:59.213224Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2507.08686"},"observation_digest":"sha256:5e65720a33c3baffc87aea25eefa4aebbfecb6c579de5c693f2f1233830ae900","observation_id":"c799125e-ab5f-436e-9539-60df4ff2888d","resolution":{"observed_at":"2026-08-06T18:19:59.213224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-06T13:25:20.610099Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.20738","last_updated":"2025-07-28T11:42:17Z","snapshot_observed_at":"2026-08-09T15:42:10.162641Z","submitted_at":"2025-07-28T11:42:17Z","title":"Dark Side of Modalities: Reinforced Multimodal Distillation for Multimodal Knowledge Graph Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T13:25:20.610099Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2507.20738"},"observation_digest":"sha256:50efc496c47f1adc12fefc03578fc34ad2b87641136c14005da68d02f7b434a9","observation_id":"70307df7-99c3-4c1e-8a12-2fd6a0d553f4","resolution":{"observed_at":"2026-08-06T13:25:20.610099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-15T17:55:13.736803Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.21182","last_updated":"2025-07-27T02:08:21Z","snapshot_observed_at":"2026-08-18T11:02:21.632028Z","submitted_at":"2025-07-27T02:08:21Z","title":"SDD: Self-Degraded Defense against Malicious Fine-tuning","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-15T17:55:13.736803Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2507.21182"},"observation_digest":"sha256:4cff9ac3c18de39559932fbb326da4f5d26a1456b622f1e04a51cdddccf2f7d2","observation_id":"5a7d25de-d0b7-4356-9b7a-ba51c792bef9","resolution":{"observed_at":"2026-08-15T17:55:13.736803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2508.00901","last_updated":"2026-05-18T13:04:24Z","snapshot_observed_at":"2026-08-18T00:30:58.619504Z","submitted_at":"2025-07-28T17:24:57Z","title":"Provable Knowledge Acquisition and Extraction in One-Layer Transformers","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T23:05:56.687644Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2508.00901"},"observation_digest":"sha256:61bfec2d4c098c44f559c8cec807157f430b28dfe125304b72256e136bf12f56","observation_id":"9263cb36-40c2-47f6-b769-3111ece9d798","resolution":{"observed_at":"2026-05-21T23:10:44.837889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-03T05:28:11.004696Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.02244","last_updated":"2026-07-14T10:18:39Z","snapshot_observed_at":"2026-08-14T23:45:08.458850Z","submitted_at":"2026-02-02T15:53:55Z","title":"Entropy-Preserving Supervised Fine-Tuning via Adaptive Self-Distillation for Large Reasoning Models","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-03T05:28:11.004696Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2602.02244"},"observation_digest":"sha256:8dcd7ec221eb5448fff38f20dbfecf609737077fb03027343bdfa6c947020fca","observation_id":"e24103e9-513a-4050-903b-797f72bea4ea","resolution":{"observed_at":"2026-08-03T05:28:11.004696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-03T04:14:12.821429Z","title":"and Li, Y","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2602.05725","last_updated":"2026-06-03T16:40:26Z","snapshot_observed_at":"2026-08-17T20:44:52.996699Z","submitted_at":"2026-02-05T14:49:40Z","title":"Muon in Associative Memory Learning: Training Dynamics and Scaling Laws","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-03T04:14:12.821429Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2602.05725"},"observation_digest":"sha256:6a0b11b2424edc8a8392699a004bb52d709dfad6b143200cc6a79392ab3cd196","observation_id":"33cf71be-8187-4fb2-a120-fce88488509f","resolution":{"observed_at":"2026-08-03T04:14:12.821429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-03T00:09:32.249407Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2602.11760","last_updated":"2026-05-28T07:31:43Z","snapshot_observed_at":"2026-08-21T02:01:01.227641Z","submitted_at":"2026-02-12T09:36:03Z","title":"Aggregate Models, Not Explanations: Improving Feature Importance Estimation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T00:09:32.249407Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2602.11760"},"observation_digest":"sha256:68235721e124c7bdcd9c6f6488ae9a95c50bfc01d5ced9b2a6831ded6f454a90","observation_id":"5eceff6e-92a1-49d2-8345-7c95fc5731ec","resolution":{"observed_at":"2026-08-03T00:09:32.249407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2604.04038","last_updated":"2026-08-17T04:11:03Z","snapshot_observed_at":"2026-08-20T23:12:25.276476Z","submitted_at":"2026-04-05T09:41:30Z","title":"FLAME: Condensing Ensemble Diversity into a Single Network for Efficient Sequential Recommendation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T17:29:06.884078Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2604.04038"},"observation_digest":"sha256:90be9aec46b831f7d47d6577a936d196956e7fa9e2fa815c78897e6d1956f52b","observation_id":"8064f121-f4e9-43bd-8976-c31d09da6d65","resolution":{"observed_at":"2026-05-13T17:33:02.560618Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2604.19724","last_updated":"2026-04-21T17:48:51Z","snapshot_observed_at":"2026-08-11T14:48:51.246220Z","submitted_at":"2026-04-21T17:48:51Z","title":"Benign Overfitting in Adversarial Training for Vision Transformers","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-10T02:58:29.672338Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2604.19724"},"observation_digest":"sha256:5a5fb2eac55a600f15da53b89788188c14fbd56caac3d67d29ed650a03d7ac65","observation_id":"9f31e33e-2a2b-4eb1-9e76-b8836b0e9461","resolution":{"observed_at":"2026-05-11T12:46:18.193794Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2605.07244","last_updated":"2026-05-08T05:01:40Z","snapshot_observed_at":"2026-08-15T21:47:26.504858Z","submitted_at":"2026-05-08T05:01:40Z","title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-11T02:02:41.411795Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2605.07244"},"observation_digest":"sha256:6bb0c2eb5c553200eab6c1270920938403abc615678be00852a18c4e6d00655f","observation_id":"c18835db-2651-443d-8d1d-4373d5fe908c","resolution":{"observed_at":"2026-05-11T04:00:55.250639Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2605.08292","last_updated":"2026-05-08T09:21:46Z","snapshot_observed_at":"2026-08-11T11:04:07.495366Z","submitted_at":"2026-05-08T09:21:46Z","title":"Hierarchical Mixture-of-Experts with Two-Stage Optimization","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-12T01:58:04.218139Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2605.08292"},"observation_digest":"sha256:198ccab477ab5e8e812eeb4cd0fd3fc350919e35b20c570046cd6279d0e63b44","observation_id":"0c46d6dd-4716-4c9c-9a79-b60780bfae17","resolution":{"observed_at":"2026-05-12T07:46:26.781487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2605.11414","last_updated":"2026-05-12T02:06:05Z","snapshot_observed_at":"2026-08-11T14:30:17.177431Z","submitted_at":"2026-05-12T02:06:05Z","title":"Generative Diffusion Prior Distillation for Long-Context Knowledge Transfer","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T02:30:57.473569Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2605.11414"},"observation_digest":"sha256:2f6d6b012fbce3564e7bb167cf8d357e3b80028497a51580f7ff496c89c8de1a","observation_id":"5355f14c-18dd-466e-8b83-980736b0a9f6","resolution":{"observed_at":"2026-05-13T02:32:06.243409Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2606.08252","last_updated":"2026-06-06T16:40:53Z","snapshot_observed_at":"2026-08-06T07:53:16.025900Z","submitted_at":"2026-06-06T16:40:53Z","title":"Quantifying and Defending against the Privacy Risk in Logit-based Federated Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T19:17:23.711817Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2606.08252"},"observation_digest":"sha256:e436430ea1823779cf9624e36cdd85991926919646238b041a2e7834d5bf1a7e","observation_id":"fbc414cf-db0b-482c-ae5f-d4d49302a57f","resolution":{"observed_at":"2026-07-02T22:07:25.932947Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2606.09658","last_updated":"2026-06-08T15:42:54Z","snapshot_observed_at":"2026-08-12T20:25:49.594664Z","submitted_at":"2026-06-08T15:42:54Z","title":"Muon Learns More Robust and Transferable Features than Adam","version":1},"reference_index":127,"source":"arxiv_source","source_observed_at":"2026-06-27T17:08:30.717799Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2606.09658"},"observation_digest":"sha256:5590e381761214eb50b0ee2b6cfa631d7d93ec4210f1263b735f8200cfeb2d24","observation_id":"312f287d-60f6-453f-89f0-e5449d3153e0","resolution":{"observed_at":"2026-07-03T00:27:30.199822Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":"2012.09816","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-04T10:29:44.312964Z","title":"Towards understanding ensemble, knowledge distillation and self-distillation in deep learning","venue":null,"work_id":"2dba5ef3-3a9d-48db-8546-a420297aede5","year":2012},"citing_paper":{"arxiv_id":"2606.23364","last_updated":"2026-06-22T14:00:26Z","snapshot_observed_at":"2026-08-02T17:37:15.368213Z","submitted_at":"2026-06-22T14:00:26Z","title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-06-26T08:53:46.285233Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2606.23364"},"observation_digest":"sha256:f4506713784df5b11fc8be845214285106ba70f0fde873eccf20c93310c9465f","observation_id":"f4fe0ad6-68e4-4bad-a815-703f5f318db0","resolution":{"observed_at":"2026-07-04T10:29:44.314748Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-11T13:53:36.775836Z","title":"arXiv preprint arXiv:2012.09816 , year=","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-07-11T13:53:36.775836Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:74bf4f8fabf2b00b92c5f424190827db501ae803d7c167a89ec39aa25dc3f80f","observation_id":"bf22cc0a-7578-4111-a104-01848e4034b1","resolution":{"observed_at":"2026-07-11T13:53:36.775836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-02T08:40:36.488598Z","title":"arXiv preprint arXiv:2012.09816 , year=","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2607.04763","last_updated":"2026-07-26T14:17:18Z","snapshot_observed_at":"2026-08-15T22:59:02.741838Z","submitted_at":"2026-07-06T07:56:53Z","title":"Multi-Turn On-Policy Distillation with Prefix Replay","version":3},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T08:40:36.488598Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2607.04763"},"observation_digest":"sha256:e352eb68a1fe7c5f462ab608491ebe457ce689419c566a0479ab7158637f33bf","observation_id":"5a86c0ce-2077-42e8-9bfc-5fe801ea1dec","resolution":{"observed_at":"2026-08-02T08:40:36.488598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-13T08:04:29.580811Z","title":"and Li, Y","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2607.08776","last_updated":"2026-05-05T07:24:01Z","snapshot_observed_at":"2026-08-18T00:26:20.893981Z","submitted_at":"2026-05-05T07:24:01Z","title":"A Unified Approach to Interpreting Knowledge Distillation for Large Language Models via Interactions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T08:04:29.580811Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2607.08776"},"observation_digest":"sha256:be7280f3f582c544c51913b40edad981b904b8d109c9a26d267d66d10b32588d","observation_id":"cbb8f483-f147-468b-89b7-a361c91ac52d","resolution":{"observed_at":"2026-07-13T08:04:29.580811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-07-31T23:44:20.615202Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23346","last_updated":"2026-07-25T19:53:58Z","snapshot_observed_at":"2026-08-20T11:09:28.471972Z","submitted_at":"2026-07-25T19:53:58Z","title":"SPRKD: Effective Knowledge Distillation for Deep Neural Networks via Saddle Region Approximation","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-07-31T23:44:20.615202Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2607.23346"},"observation_digest":"sha256:8c229932d096b3efa1a271af8baa93ae7055dbe0257fd279aed4bdaaa32368ce","observation_id":"45eca94a-1728-4cda-9c6c-3face8a7e4c1","resolution":{"observed_at":"2026-07-31T23:44:20.615202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.09816","snapshot_observed_at":"2026-08-12T00:48:45.240771Z","title":"arXiv preprint arXiv:2012.09816 , year=","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2608.07870","last_updated":"2026-08-08T02:44:43Z","snapshot_observed_at":"2026-08-15T15:49:08.325082Z","submitted_at":"2026-08-08T02:44:43Z","title":"V-Simba: Unleashing the Architectural Potential of RL in Visual Continuous Control","version":1},"reference_index":142,"source":"arxiv_source","source_observed_at":"2026-08-12T00:48:45.240771Z"},"links":{"cited_paper":"/paper/2012.09816","citing_paper":"/paper/2608.07870"},"observation_digest":"sha256:7bc8c5bdd084750e09c4239f23ed71cc1808ea5c76bf0704299099ec1baffe43","observation_id":"037584ed-01d1-49c4-9ca5-695efc6b310c","resolution":{"observed_at":"2026-08-12T00:48:45.240771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2012.09816/citation-record","integrity":"/paper/2012.09816/integrity","json":"/paper/2012.09816/citation-record.json","paper":"/paper/2012.09816"},"outbound":[],"paper":{"arxiv_id":"2012.09816","last_updated":"2023-02-15T10:01:31Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-20T11:06:19.455382Z","submitted_at":"2020-12-17T18:34:45Z","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 38 inbound Pith citation observations for arXiv:2012.09816."}