{"as_of":"2026-08-14T03:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:33f6285d6c03ab0b3edf89201a09c3751a93cf92870d914cae5e2fd6a78bb290","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:27:41.962913Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.15133/citation-record","integrity":"/paper/2505.15133/integrity","json":"/paper/2505.15133/citation-record.json","paper":"/paper/2505.15133"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-07T15:27:36.503841Z","title":"Distilling the knowledge in a neural network,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.503841Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:72930be90767a8ea38bd515f00a3d94e735a5d354bf755834395139308573e16","observation_id":"e0d404fe-84a1-4882-bc5f-aa7ea4bf6621","resolution":{"observed_at":"2026-08-07T15:27:36.503841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:46.575560Z","title":"Pkd: General distillation framework for object detectors via pearson correlation coefficient,","venue":null,"work_id":"f10ec435-0533-49f4-9f8c-0ffc529733fe","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.578218Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:b6190041b995648f3555aa42cfc3cef96d41e02ce56bab7ceb8fd3135b181a4e","observation_id":"64b2b912-3b68-4659-89e3-4725851ac93f","resolution":{"observed_at":"2026-08-07T15:27:46.663089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:46.443290Z","title":"Cross-image relational knowledge distillation for semantic segmentation,","venue":null,"work_id":"35ae8dcd-205e-4e0e-93fa-421d51eb9f28","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.654784Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:e013318f9215351b21b22eb5f5bf07925a5b8c0b7183cc623f5b5edf65e80387","observation_id":"467201bb-2142-498f-b62a-937a8ad9c233","resolution":{"observed_at":"2026-08-07T15:27:46.492888Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:46.303362Z","title":"Relational diffusion distillation for efficient image generation,","venue":null,"work_id":"3acb51c9-5f52-4924-b536-b6a3c81c53de","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.732477Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:6e87dd37c2e93224bb49e01b484b36a5c1beb8550d71017709bc3203a4dd58d6","observation_id":"048105f7-9498-482e-92b3-a565ffb73c11","resolution":{"observed_at":"2026-08-07T15:27:46.383344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:36.856590Z","title":"Knowledge distillation: A survey,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.856590Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:ecc26d48704190ffc2469e22270132421158ea7de88da8759acf88fdbdeeab9c","observation_id":"5092153c-9184-4b2f-8096-1e0113b51c65","resolution":{"observed_at":"2026-08-07T15:27:36.856590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:46.184918Z","title":"Knowledge distillation from single-task teachers to multi-task student for end-to-end au- tonomous driving,","venue":null,"work_id":"8cae5fee-d8ad-4aee-a273-abf6185d9aaa","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:36.954959Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:2168cd24db8df3c54dd957d0bbc5768aa793e3f0558fbcfb0feaa0b2d472cea8","observation_id":"41b20559-7d8e-479e-a778-ad088e418074","resolution":{"observed_at":"2026-08-07T15:27:46.217421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.04619","last_updated":"2024-04-06T12:51:00Z","snapshot_observed_at":"2026-08-13T00:35:56.450563Z","submitted_at":"2024-04-06T12:51:00Z","title":"Do We Really Need a Complex Agent System? Distill Embodied Agent into a Single Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.04619","snapshot_observed_at":"2026-08-07T15:27:37.033544Z","title":"Do we really need a complex agent system? distill embodied agent into a single model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.033544Z"},"links":{"cited_paper":"/paper/2404.04619","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:cbb7a8a147e7435f10beffaeec7dab7982ace70e69fd7e34bd92d4c353beaa8a","observation_id":"b5d56058-ea41-4dab-81a0-663b44a4f22f","resolution":{"observed_at":"2026-08-07T15:27:37.033544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01620","last_updated":"2024-06-07T18:13:16Z","snapshot_observed_at":"2026-08-14T00:59:30.001024Z","submitted_at":"2024-02-02T18:35:14Z","title":"MAGDi: Structured Distillation of Multi-Agent Interaction Graphs Improves Reasoning in Smaller Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01620","snapshot_observed_at":"2026-08-07T15:27:37.106377Z","title":"Magdi: Structured distillation of multi-agent interaction graphs improves reasoning in smaller language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.106377Z"},"links":{"cited_paper":"/paper/2402.01620","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:a5e3da5d1af3e132f4225da2c37031d0f002044d039c1b2926faeb0962b9bbdf","observation_id":"87b6d3ed-9c3b-4d10-ac30-0637744beb7c","resolution":{"observed_at":"2026-08-07T15:27:37.106377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:46.033322Z","title":"Adaptive multi-teacher knowledge distillation with meta-learning,","venue":null,"work_id":"bcae2691-2392-4f8c-a65a-13a156e93b9c","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.182448Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:18296a094e1110d9ec409e337f16c288b5ad3da0d66b6772031be1fec944a435","observation_id":"6166566b-e677-426c-9f83-500f7b089d71","resolution":{"observed_at":"2026-08-07T15:27:46.092176Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.966826Z","title":"Multi-teacher knowledge distillation with reinforcement learning for visual recognition,","venue":null,"work_id":"27426c42-7c0b-4e3f-8df8-7b535acd290c","year":2025},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.261794Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:7fde1982b2c8f6bde5ee27af260b5455d37b48c7c62e8ecc085e14fe61394e1e","observation_id":"f1a44d36-16d0-46d4-9421-234512dbad1f","resolution":{"observed_at":"2026-08-07T15:27:45.996742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.810951Z","title":"A comprehensive overhaul of feature distillation,","venue":null,"work_id":"1cc15c0b-4553-46b0-947b-aac9375f3f15","year":2019},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.354357Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:4ae906fcaf84bd3db6b318df04f6dcea25c74526034e41c95d2e4e9b9bc3c935","observation_id":"6608e0ae-61ac-461a-b50f-01971e9db70f","resolution":{"observed_at":"2026-08-07T15:27:45.912751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.659885Z","title":"What makes a","venue":null,"work_id":"046f6c88-1ae8-4ca4-8e27-31ae5b0a7426","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.459039Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:7d6ff58e328b76f0f9b5b01f2578e8e5064aaad570667e5bc5a6ef7ae6339367","observation_id":"b792ffef-1906-4fcb-bad1-7f0d4f376713","resolution":{"observed_at":"2026-08-07T15:27:45.733883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.508210Z","title":"Cross-view consistency regularisation for knowledge distillation,","venue":null,"work_id":"cb40ce66-8a01-4c48-8c6b-91cd7665cf00","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.532823Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:bf6f2f1785425779b633d61d5405267868f610cd92db4dfbc7bc0d448aa003f0","observation_id":"6399e879-f0b9-4fcb-b5dd-14a752be7a78","resolution":{"observed_at":"2026-08-07T15:27:45.553511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.361748Z","title":"Why logit distillation works: A novel knowledge distillation technique by deriving target augmentation and logits distortion,","venue":null,"work_id":"40ef7b90-390b-4394-a10f-ccb840a7d9f7","year":2025},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.623360Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:80c43b2c8714992c2a89fe98271062c84d287a8021f9162bf96b09905d074704","observation_id":"d032f3f7-5560-4133-8105-236682a551ad","resolution":{"observed_at":"2026-08-07T15:27:45.448812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.231667Z","title":"Single teacher, multiple perspectives: Teacher knowledge augmentation for enhanced knowledge distillation,","venue":null,"work_id":"2e563c41-9550-40b4-8c3b-855db1cdefea","year":2025},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.724296Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:08fc5d6b4a673490149fafd63d0cc0a5624be94ce6192c124a307da552bf4b2f","observation_id":"ca360851-f8b1-4ea1-adf6-5a265f7dadd9","resolution":{"observed_at":"2026-08-07T15:27:45.271202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:45.129810Z","title":"Revisiting knowledge distillation via label smoothing regularization,","venue":null,"work_id":"e8bc7cd8-75f6-46a1-ba66-17fb40cf2c0d","year":2020},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.806925Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:b62fbd9c356fd0ecc4b44fe5629bbad22d7fd7621605411be5c1469af4d082a1","observation_id":"c0201d6e-27ff-4f44-b48d-3af497d56479","resolution":{"observed_at":"2026-08-07T15:27:45.174276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.970788Z","title":"Debiased distillation for consistency regularization,","venue":null,"work_id":"fa7fa9d0-d4ec-4311-bfc4-439de11700a8","year":2025},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.893043Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:f0a782d0609063e7b1149ae24d9f71cf3bf843b7907c0ffd8501d34db9430fb5","observation_id":"63ba9aac-2e8f-4535-9ab1-63c20a700ab0","resolution":{"observed_at":"2026-08-07T15:27:45.010488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:37.975545Z","title":"Decoupled knowledge distillation,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:37.975545Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:e228282f65001b358e11e45ecd0f1dbe85612aa5c6ac999628e71e15035f2eaf","observation_id":"ac41c58b-b66a-496b-9ce1-85c355043b37","resolution":{"observed_at":"2026-08-07T15:27:37.975545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.848364Z","title":"Dot: A distillation-oriented trainer,","venue":null,"work_id":"7c132cd8-9158-4460-ab61-6def5515461c","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.097181Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:ed077b5f7034a1f4c306f6f60e876d3c6ea998fb8c7d03b7ccc19cb455b0cf13","observation_id":"6397544c-5d4c-460d-b1f0-97864a93412a","resolution":{"observed_at":"2026-08-07T15:27:44.924995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.731994Z","title":"Medhi, Stochastic processes","venue":null,"work_id":"db468536-ff5a-4f92-808d-7455ba25ae99","year":1994},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.183381Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:8e88ee67bd9d14a60d19fff721a62a270f2a5c633cb7ab0c8d0e218a38d9766b","observation_id":"88a4b7fa-7ffe-4c10-89c5-3a9e18616648","resolution":{"observed_at":"2026-08-07T15:27:44.770014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.07384","last_updated":"2020-02-24T10:47:39Z","snapshot_observed_at":"2026-08-12T10:23:57.533405Z","submitted_at":"2020-01-21T08:33:29Z","title":"Understanding Why Neural Networks Generalize Well Through GSNR of Parameters","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.07384","snapshot_observed_at":"2026-08-07T15:27:38.291896Z","title":"Understanding why neural networks generalize well through gsnr of parameters,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.291896Z"},"links":{"cited_paper":"/paper/2001.07384","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:7fd3d18b8c83e91482a511c702150070dd7b7d175bb81f0bd1a4e8d0161f410b","observation_id":"61a9c61d-76fa-4573-b50b-969855e3b1e8","resolution":{"observed_at":"2026-08-07T15:27:38.291896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.603232Z","title":"Towards understanding how momentum improves generalization in deep learning,","venue":null,"work_id":"e8fe6e9a-b37f-44e6-9ace-b5200a32d0a5","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.372172Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:f50b3f172500120efd901ccc2d45b737d2d4e191f7f3ce3845e9436ae56c467e","observation_id":"12ca687c-de2b-400b-92d1-d57d5b2d29b7","resolution":{"observed_at":"2026-08-07T15:27:44.684333Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.458006Z","title":"Visualizing the loss landscape of neural nets,","venue":null,"work_id":"cf605137-f854-4360-b4fc-2b0199307a9d","year":2018},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.452250Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:2fd8f3c2216083c3e896c2ec3dc752c61f1ab64fcaa3533c461fc60e5a44ca3a","observation_id":"e7d7e849-6faf-432d-8b4b-3ab51de783b8","resolution":{"observed_at":"2026-08-07T15:27:44.516004Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.336065Z","title":"Tighter variational bounds are not necessarily better,","venue":null,"work_id":"88e660fd-1386-44c6-b877-9127331d4904","year":2018},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.562746Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:d1710c46fb73a548ea2fc5cdbde85ca4569b4caf7e56fca75bc0db6011485055","observation_id":"04c92e22-e759-4f8e-9f62-40a26f7b97a2","resolution":{"observed_at":"2026-08-07T15:27:44.422216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.04836","last_updated":"2017-02-09T20:38:16Z","snapshot_observed_at":"2026-07-06T05:10:58.923264Z","submitted_at":"2016-09-15T20:03:06Z","title":"On Large-Batch Training for Deep Learning: Generalization Gap and Sharp Minima","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.04836","snapshot_observed_at":"2026-08-07T15:27:38.643965Z","title":"On large-batch training for deep learning: Generalization gap and sharp minima,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.643965Z"},"links":{"cited_paper":"/paper/1609.04836","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:45ec6bb2eb96023bdd879d0f978a76cb4201f7d21163d10eac809252d75999ce","observation_id":"6da10cd4-e92a-464c-b70a-efd994502bdc","resolution":{"observed_at":"2026-08-07T15:27:38.643965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05407","last_updated":"2019-02-25T14:18:11Z","snapshot_observed_at":"2026-07-31T19:09:21.589864Z","submitted_at":"2018-03-14T17:09:27Z","title":"Averaging Weights Leads to Wider Optima and Better Generalization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05407","snapshot_observed_at":"2026-08-07T15:27:38.742449Z","title":"Averaging weights leads to wider optima and better generalization,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.742449Z"},"links":{"cited_paper":"/paper/1803.05407","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:58b0a7638bb4620321e5aefebff972dccd85480ca7bca34d1030ed7287644d70","observation_id":"5afee9b5-69c2-40db-9483-4f2cb568233f","resolution":{"observed_at":"2026-08-07T15:27:38.742449Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:38.796907Z","title":"Curriculum learning,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.796907Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:40873b899747413b9d7c52ce719c3ecd3ec520c1e4b1655b9166d4a0dec111ed","observation_id":"f5091cf0-be73-4ccf-bcaf-f3443446fc1c","resolution":{"observed_at":"2026-08-07T15:27:38.796907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.216752Z","title":"Curriculum temperature for knowledge distillation,","venue":null,"work_id":"d63bb96a-5185-40dd-a8cc-efb11982d3d9","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:38.919632Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:006e81076037fe1665f05b081aac8aa624d12d82e934fb98d446d7709c605bbd","observation_id":"24e1d452-a394-4b57-9a9b-0e5825fd2bba","resolution":{"observed_at":"2026-08-07T15:27:44.251571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:44.089808Z","title":"Improving knowledge distillation via head and tail categories,","venue":null,"work_id":"dcdb80d3-74ca-4522-af9d-ed8f86ae13a2","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.021002Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:70ce071fe9cfdf0b258e0958942e1f454485c1c579c52a8d21061f1b0320a302","observation_id":"c3a8e71d-4001-4eae-9d9b-84fe6d0e6c0f","resolution":{"observed_at":"2026-08-07T15:27:44.147153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6550","last_updated":"2015-03-27T11:52:28Z","snapshot_observed_at":"2026-07-06T04:04:16.777653Z","submitted_at":"2014-12-19T22:40:51Z","title":"FitNets: Hints for Thin Deep Nets","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6550","snapshot_observed_at":"2026-08-07T15:27:39.103516Z","title":"Fitnets: Hints for thin deep nets,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.103516Z"},"links":{"cited_paper":"/paper/1412.6550","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:73906e2a380d7a9437858bd111257a7dc7d0b772051752dc29bf6f359e869230","observation_id":"81c3a00c-306c-4316-b095-7ac8416df457","resolution":{"observed_at":"2026-08-07T15:27:39.103516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.842349Z","title":"From knowledge distillation to self-knowledge distillation: A unified approach with normalized loss and customized soft labels,","venue":null,"work_id":"53c79195-64c5-4e0b-ab66-27bcd00eeddc","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.485521Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:121c6ce89152cfcd68cd6f4766597631afb874f1cff3e6245a38943e73287564","observation_id":"3a78a949-c106-4b62-815a-c464be8542d5","resolution":{"observed_at":"2026-08-07T15:27:43.913091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.07485","last_updated":"2021-01-12T08:46:08Z","snapshot_observed_at":"2026-08-02T18:55:29.830671Z","submitted_at":"2020-10-15T03:03:36Z","title":"Reducing the Teacher-Student Gap via Spherical Knowledge Disitllation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.07485","snapshot_observed_at":"2026-08-07T15:27:39.597718Z","title":"Reducing the teacher-student gap via spherical knowledge disitllation,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.597718Z"},"links":{"cited_paper":"/paper/2010.07485","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:d4a7cce69ab893b80a5f22a0b1440211725946e556001ba1a46de5c11cb3fbb0","observation_id":"d62e952e-6251-41cf-878c-757fde8dec17","resolution":{"observed_at":"2026-08-07T15:27:39.597718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.734537Z","title":"Mdr: Multi-stage decoupled relational knowledge distillation with adaptive stage selection,","venue":null,"work_id":"12e6f39d-eae9-46a8-a5cd-d1fb70393ad9","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.716428Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:a8e54d90b4acee0daa4b7b3a88fc1c2abbbc9208c5439d2e3090b09aca885cb3","observation_id":"67eb321e-2747-41ad-aa17-bfc5e4ce8bb5","resolution":{"observed_at":"2026-08-07T15:27:43.774648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.623421Z","title":"Ntce-kd: Non-target-class-enhanced knowledge distillation,","venue":null,"work_id":"69848abf-e73b-4f3e-b2ab-209bc178c0fb","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.795288Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:633e66492f94f4d2d3777df93e40e3fb9f5fe3f5738e13ff439f09ac143bd402","observation_id":"d5dc53a5-1c2b-492d-842c-c2f633ac00a7","resolution":{"observed_at":"2026-08-07T15:27:43.669628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.496367Z","title":"Teach less, learn more: On the undistillable classes in knowledge distillation,","venue":null,"work_id":"7e028cb9-a99c-4a70-8320-a41d4e5ccab5","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.874730Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:dabe2cb3586931625a50086f776e59f3b968be9c163da62642f789f431929395","observation_id":"d2f9a4b5-dbe6-4d8c-b83c-5845dfaa0bb4","resolution":{"observed_at":"2026-08-07T15:27:43.554923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07703","last_updated":"2025-07-27T12:26:24Z","snapshot_observed_at":"2026-08-13T01:23:11.621752Z","submitted_at":"2024-08-14T17:59:32Z","title":"Knowledge Distillation with Refined Logits","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07703","snapshot_observed_at":"2026-08-07T15:27:39.980502Z","title":"Knowledge distillation with refined logits,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:39.980502Z"},"links":{"cited_paper":"/paper/2408.07703","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:d8d1f3a5350990966bf135b4d664904f627e44b6757d083fe2095c2497ae7262","observation_id":"3764e83b-aff0-4d4a-9bd1-e7eb7eb6d295","resolution":{"observed_at":"2026-08-07T15:27:39.980502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.403458Z","title":"Domain generalization guided by gradient signal to noise ratio of parameters,","venue":null,"work_id":"5f946bb4-eccf-423c-a0fb-cd2763a572a2","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.074252Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:55271449fb1f60e703769ff1f3bb336461f7fbe5ebc098a3d67027ff283e7669","observation_id":"af30dbb8-8622-40bb-a948-fa405798e00a","resolution":{"observed_at":"2026-08-07T15:27:43.449455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.294154Z","title":"On the importance of initialization and momentum in deep learning,","venue":null,"work_id":"6aab1b3e-fa1e-4993-82e8-f255731404b5","year":2013},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.179286Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:e535415ac55af1c8f7142efd0b5fd7ba151879d522c6a6c876c2a3dea5ef95a1","observation_id":"e2ea07f6-45e4-46f9-a864-dd2202ede79e","resolution":{"observed_at":"2026-08-07T15:27:43.332392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-07T15:27:40.246422Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.246422Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:1665b088390ba4ac135989214fbd7620bbf96a43c21eab11148b0cb6d3f1a1b5","observation_id":"a1ec5fe2-5abf-4729-b67e-2050f1fe33da","resolution":{"observed_at":"2026-08-07T15:27:40.246422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.183547Z","title":"Training data-efficient image transformers & distillation through attention,","venue":null,"work_id":"dd287470-bcf0-4207-a855-5fc593cf4d78","year":2021},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.299744Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:36ee2a9dd8c566a79638f5ee540d9beb96a01555f9b0dd04e4401ab64cda3879","observation_id":"d69f7d98-70f0-4dab-8bc1-e997aa6b81c6","resolution":{"observed_at":"2026-08-07T15:27:43.224578Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.075432Z","title":"Logit standardization in knowledge distillation,","venue":null,"work_id":"59795512-fd7f-4749-abcf-30bfef2d9f1c","year":2024},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.439946Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:ce97528ae52ddee21600813193f5b50c9ef8c225c3846abb1cc1a8321dbfa270","observation_id":"bf2ac38d-8b05-43ea-8c90-8b6dd8a6a301","resolution":{"observed_at":"2026-08-07T15:27:43.130329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:40.599509Z","title":"Learning multiple layers of features from tiny images,","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.599509Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:eb7ccee65d7e621653fe8cf98ad153fcde51e527be611eaea842dc88d4b3e9a2","observation_id":"aec4fc97-6318-45a5-a673-7d6192ad26dc","resolution":{"observed_at":"2026-08-07T15:27:40.599509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.972427Z","title":"ImageNet large scale visual recognition challenge,","venue":null,"work_id":"0f561ddc-9b62-4729-919e-4051239efd61","year":2015},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.736777Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:fd92ce82d59382e07e6428c55b96c9dbb05a73b9ea93146e262a595036a855d9","observation_id":"1fdf1eb6-8e30-4f23-b400-e44c51576af0","resolution":{"observed_at":"2026-08-07T15:27:43.016295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.881592Z","title":"Microsoft coco: Common objects in context,","venue":null,"work_id":"68ea5437-d6a3-4016-9bc4-9abbb41245ef","year":2014},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.848369Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:784e898470f742bbb13266166a7ed4d3a05db3dc4ffc9c94ef443c64230a2fe9","observation_id":"faaeaf93-e5fb-4ba9-a74f-ed0458292694","resolution":{"observed_at":"2026-08-07T15:27:42.929092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.786636Z","title":"Knowledge distillation with the reused teacher classifier,","venue":null,"work_id":"3e28b6fd-d0c0-4edc-bb40-7b0452022102","year":2022},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:40.978509Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:94a95f612ed6a7d9a9c96f78ee6d7fe9159df99c6c715d7c0192c375e43d3526","observation_id":"c09611ed-bad6-4b6d-b5a6-d0fedbb3776c","resolution":{"observed_at":"2026-08-07T15:27:42.824884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.725387Z","title":"Class attention transfer based knowledge distillation,","venue":null,"work_id":"30453656-48ae-4fe9-9b01-d914cb7372d8","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.094829Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:4ce6963309d2b28dd5bfc4ef5c36a0f6bb66c6fbd981ab1c579a063f62cf6628","observation_id":"fb03f528-b482-4d0d-a446-a7d10a963463","resolution":{"observed_at":"2026-08-07T15:27:42.756410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.628261Z","title":"Multi-level logit distillation,","venue":null,"work_id":"46782e71-1489-4e34-98e8-2e809e9198e7","year":2023},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.235257Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:e7f4357db596218dd74a9651f193ce214b453009d5c7af3846e53f2c88754fce","observation_id":"4c520fb1-b70b-4416-b9ca-e4e543f245b9","resolution":{"observed_at":"2026-08-07T15:27:42.673550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:43.969915Z","title":"Distilling knowledge via knowledge review,","venue":null,"work_id":"7c347d34-f038-465a-86eb-c96e3f947093","year":2021},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.348891Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:c6cd157c8d0764264fcd5fadf4520f0253a4800f09c112805ec67ceb01b428d0","observation_id":"2cfb5423-4942-4a95-94d2-7cb2966d4faa","resolution":{"observed_at":"2026-08-07T15:27:44.027869Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.499954Z","title":"Faster r-cnn: Towards real-time object detection with region proposal networks,","venue":null,"work_id":"7ec406e7-88f4-4ce5-b3cb-90dce7101056","year":2015},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.441181Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:26b28a735a84cc5daff4f3bc6e79c7453f12ae4bc1c99c20514a0aff62ad20d6","observation_id":"b06a12d0-3a00-44c5-bad6-4897748c8ba2","resolution":{"observed_at":"2026-08-07T15:27:42.559368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.337244Z","title":"Visualizing data using t-sne","venue":null,"work_id":"59564005-c8b6-4871-8c80-60d4c21de659","year":2008},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.607601Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:56b15141cdb9457e573f7bbb255d4f64a2a99f252c9bcc3526342cf38f4192b6","observation_id":"e559f72b-9c14-4c70-8994-fe8a9f1feda7","resolution":{"observed_at":"2026-08-07T15:27:42.423973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.03928","last_updated":"2017-02-12T22:05:47Z","snapshot_observed_at":"2026-08-10T12:33:31.960191Z","submitted_at":"2016-12-12T21:15:57Z","title":"Paying More Attention to Attention: Improving the Performance of Convolutional Neural Networks via Attention Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.03928","snapshot_observed_at":"2026-08-07T15:27:41.661879Z","title":"Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.661879Z"},"links":{"cited_paper":"/paper/1612.03928","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:84d8f3c24ebda8687637496c1b6cd0b87f34a3b097d71d216d7531678ea56be0","observation_id":"d36ba7e0-0db2-47a6-8d66-694eaae47c12","resolution":{"observed_at":"2026-08-07T15:27:41.661879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:41.791141Z","title":"Relational knowledge distillation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.791141Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:7d43df9e6c7bef4e68f8264a194f5673870c4b94038034dd7be95082473dbfef","observation_id":"1c1127a6-305c-42a6-b0a5-c75c7b0fa479","resolution":{"observed_at":"2026-08-07T15:27:41.791141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.10699","last_updated":"2022-01-24T19:12:34Z","snapshot_observed_at":"2026-08-10T18:58:30.893017Z","submitted_at":"2019-10-23T17:59:18Z","title":"Contrastive Representation Distillation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.10699","snapshot_observed_at":"2026-08-07T15:27:41.869574Z","title":"Contrastive representation distillation,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.869574Z"},"links":{"cited_paper":"/paper/1910.10699","citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:3c578c194228c620bf5f76e79e7c48f27ce6484ab0944ff67e5bc186a3cad988","observation_id":"a3818fe6-56fc-416c-b67f-1a76cca1fc33","resolution":{"observed_at":"2026-08-07T15:27:41.869574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:27:42.193331Z","title":"A comprehensive overhaul of feature distillation,","venue":null,"work_id":"276bd000-55d5-485d-8d7d-7bc7ad05212c","year":2019},"citing_paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T15:27:41.962913Z"},"links":{"citing_paper":"/paper/2505.15133"},"observation_digest":"sha256:9e3ddeb6c5396f2f6ab25a7a49740c3a349f2f4c95bc0b4f4f281f5be6135e15","observation_id":"902d1831-01c9-401f-82a4-cdd4acd83bf9","resolution":{"observed_at":"2026-08-07T15:27:42.274448Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.15133","last_updated":"2025-05-21T05:38:57Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-12T16:11:56.521177Z","submitted_at":"2025-05-21T05:38:57Z","title":"DeepKD: A Deeply Decoupled and Denoised Knowledge Distillation Trainer"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":37},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2505.15133."}