{"as_of":"2026-08-14T02:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0a5674229e03a8648c87433c77b8b40d9de4f6b1373e045eb317e15450a72840","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T17:49:20.124168Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.05837/citation-record","integrity":"/paper/2502.05837/integrity","json":"/paper/2502.05837/citation-record.json","paper":"/paper/2502.05837"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.446919Z","title":"Wav2Vec 2.0: A framework for self- supervised learning of speech representations,","venue":null,"work_id":"8bee3a5d-58d4-43cd-8cd2-81cd924c30c3","year":2020},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.025322Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:79d3f6ebfe97f462f809e2232f70831b32e44a5654bb5bddb679d2e10447f012","observation_id":"aa80bbbe-00dc-4db5-90a7-452887523ed5","resolution":{"observed_at":"2026-08-08T17:49:20.450704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.436731Z","title":"HuBERT: Self-supervised speech represen- tation learning by masked prediction of hidden units,","venue":null,"work_id":"bd1e64bf-a8fd-485d-b116-8e02f75e2ac1","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.029194Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:edbf549baa0c94444a28ecd7492747731fe0cf6539f1351b11b22be669438830","observation_id":"8959c072-4045-4e2d-aafb-a71433250fa7","resolution":{"observed_at":"2026-08-08T17:49:20.440376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.426623Z","title":"WavLM: Large-scale self-supervised pre- training for full stack speech processing,","venue":null,"work_id":"6815951d-1499-48f0-a450-502186088438","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.032936Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:6a2c975fe9124c1a028cf9a84d2eac501e92bf0841a1f4673d0e42ddae2d25b7","observation_id":"8c5c7bd1-84fa-4f6d-b34e-a4611a78b485","resolution":{"observed_at":"2026-08-08T17:49:20.430276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.415980Z","title":"Data2Vec: A general framework for self- supervised learning in speech, vision and language,","venue":null,"work_id":"be530765-ab6a-4549-ba19-891d2bc5cb18","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.036484Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:d8feb8496135c5ec4d911ad4116be49b4e1270ac1513ed58e3a89ad5cbe905d3","observation_id":"c7ade204-9fce-490a-b85f-85b4c209f8cd","resolution":{"observed_at":"2026-08-08T17:49:20.419705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.406309Z","title":"DistilHuBERT: Speech representation learning by layer-wise distillation of hidden-unit BERT,","venue":null,"work_id":"db2b383d-00be-4cf0-b8fe-f1095c5be32f","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.040170Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:9531895776f7984a6b35ac0131fc819cfee62eac218bbce376c7f37f1c909c52","observation_id":"9c55c28f-ce28-4bbf-93d7-da31612ccb5d","resolution":{"observed_at":"2026-08-08T17:49:20.409655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1503.02531","last_updated":"2015-03-09T15:44:49Z","snapshot_observed_at":"2026-07-06T04:11:24.157003Z","submitted_at":"2015-03-09T15:44:49Z","title":"Distilling the Knowledge in a Neural Network","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1503.02531","snapshot_observed_at":"2026-08-08T17:49:20.043583Z","title":"Distilling the knowledge in a neural network,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.043583Z"},"links":{"cited_paper":"/paper/1503.02531","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:35fa22139f7da37ad7f48f5f830971e0d14745c8fb4c4f19cd2692513e175990","observation_id":"a1fbe231-f647-447f-a0ce-6c0f9eaa4403","resolution":{"observed_at":"2026-08-08T17:49:20.043583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09842","last_updated":"2023-12-15T14:48:14Z","snapshot_observed_at":"2026-08-13T05:00:55.743159Z","submitted_at":"2023-12-15T14:48:14Z","title":"On the compression of shallow non-causal ASR models using knowledge distillation and tied-and-reduced decoder for low-latency on-device speech recognition","version":1},"cited_work":{"arxiv_id":"2312.09842","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.09842","snapshot_observed_at":"2026-08-08T17:49:20.259039Z","title":"On the compression of shallow non-causal ASR models using knowledge distillation and tied-and-reduced decoder for low-latency on-device speech recognition","venue":"cs.SD","work_id":"bb58e5ca-6329-4613-ad5e-60dfb51830f8","year":2023},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.047538Z"},"links":{"cited_paper":"/paper/2312.09842","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:daade61dce446b0c655a0d3a9de453ff83f74d636978e1feef6d712805a6dcd0","observation_id":"36ff72e8-8318-4d5c-a8e5-853e4c65dd04","resolution":{"observed_at":"2026-08-08T17:49:20.264126Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.396893Z","title":"FitHuBERT: Going thinner and deeper for knowledge distillation of speech self-supervised models,","venue":null,"work_id":"439dc762-db61-44b5-9b1b-51277f19e67b","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.050959Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:6ccc98dda5c7a471ead6b988d87c51b8d4eaf19ef895116eb23bb3e7e822cd03","observation_id":"d652d0ed-beb4-4c90-9946-441530e39e25","resolution":{"observed_at":"2026-08-08T17:49:20.400359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.387152Z","title":"Multi-stage progressive compression of conformer transducer for on-device speech recognition.,","venue":null,"work_id":"433d49a2-63f7-498c-98b1-b9db782733c1","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.054378Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:f0bee069a664f39778d6b4c25f0996b331dc8db5d178da2e7f5da8afb7537dcd","observation_id":"d04510a7-9017-40bf-ac00-004ceea154c1","resolution":{"observed_at":"2026-08-08T17:49:20.390810Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.04732","last_updated":"2021-03-28T19:04:25Z","snapshot_observed_at":"2026-08-14T00:39:23.744952Z","submitted_at":"2019-10-10T17:44:18Z","title":"Structured Pruning of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.04732","snapshot_observed_at":"2026-08-08T17:49:20.057508Z","title":"Struc- tured pruning of large language models,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.057508Z"},"links":{"cited_paper":"/paper/1910.04732","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:cbdc349e00ed5302f360b018778bd43d00e838017e5563338ae31bfb24f108d2","observation_id":"98387921-dfc8-4f5b-9c4f-a28fd10d8c91","resolution":{"observed_at":"2026-08-08T17:49:20.057508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.01312","last_updated":"2018-06-22T14:54:59Z","snapshot_observed_at":"2026-08-12T23:03:36.254662Z","submitted_at":"2017-12-04T19:20:27Z","title":"Learning Sparse Neural Networks through $L_0$ Regularization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.01312","snapshot_observed_at":"2026-08-08T17:49:20.061102Z","title":"Learning sparse neural networks through l 0 regulariza- tion,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.061102Z"},"links":{"cited_paper":"/paper/1712.01312","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:44d4648d950b29a039ef475426d0844a4c9cdc6a350201355030f445d7e70728","observation_id":"4aa692a9-5315-4358-afea-39d45fb6e2e1","resolution":{"observed_at":"2026-08-08T17:49:20.061102Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.376708Z","title":"PARP: Prune, Ajust and Re-Prune for self- supervised speech recognition,","venue":null,"work_id":"2c7c36e1-4118-43db-b1f6-915c448c3419","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.064745Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:f536076af3239807ada7161237a5a78351eb5420d5d090d32ab5e29565e66a1c","observation_id":"1eee6c9d-7237-4aa0-a4ac-69c1facd3cd4","resolution":{"observed_at":"2026-08-08T17:49:20.380487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.365930Z","title":"Unstructured Pruning and Low Rank Factorisation of self-supervised pre-trained speech models,","venue":null,"work_id":"d7357257-5ffe-415c-92da-020c05d7c4d0","year":2024},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.067856Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:e80902b236c5fc2443f6cde9b840090350dda8c77a4cae3c38411081b1bb553b","observation_id":"7bd7e06b-c6aa-46d0-bb3c-5161e12ece1c","resolution":{"observed_at":"2026-08-08T17:49:20.369680Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17651","last_updated":"2023-05-28T07:09:33Z","snapshot_observed_at":"2026-08-13T11:30:56.928331Z","submitted_at":"2023-05-28T07:09:33Z","title":"DPHuBERT: Joint Distillation and Pruning of Self-Supervised Speech Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.17651","snapshot_observed_at":"2026-08-08T17:49:20.070978Z","title":"DPHuBERT: Joint distillation and prun- ing of self-supervised speech models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.070978Z"},"links":{"cited_paper":"/paper/2305.17651","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:8b9614cd5cf619fc925bc99ac8cc21988e7f76bab9b9d71b72c196a1a7f0a88c","observation_id":"18b27de1-b666-4bfc-9a39-efa3349806c9","resolution":{"observed_at":"2026-08-08T17:49:20.070978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.354273Z","title":"Deep versus Wide: An analysis of student architectures for task-agnostic knowledge distil- lation of self-supervised speech models,","venue":null,"work_id":"56a9f0d2-3913-437f-82c5-07a8af045696","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.074439Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:6d10111131d0ca4a93bb689156fabe8c31c3cc67096ec1f2b2c4113dbb4be532","observation_id":"6217aa36-e406-478b-91eb-be941304d6c4","resolution":{"observed_at":"2026-08-08T17:49:20.358606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.09791","last_updated":"2020-04-29T20:49:05Z","snapshot_observed_at":"2026-08-11T01:59:17.823546Z","submitted_at":"2019-08-26T16:46:23Z","title":"Once-for-All: Train One Network and Specialize it for Efficient Deployment","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.09791","snapshot_observed_at":"2026-08-08T17:49:20.077724Z","title":"Once for all: Train one network and specialize it for efficient deployment,","venue":null,"work_id":null,"year":1908},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.077724Z"},"links":{"cited_paper":"/paper/1908.09791","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:450092b657daf7e3c1302ffb2616b0990331799d77437f70c1d0e086096b8124","observation_id":"02f74818-fb6d-4c7e-b14a-a15d7ed7e395","resolution":{"observed_at":"2026-08-08T17:49:20.077724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.344636Z","title":"LightHuBERT: Lightweight and config- urable speech representation learning with once-for-all hidden-unit BERT,","venue":null,"work_id":"747eadfc-9d4b-47a7-b78f-252fb5bfc87d","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.081322Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:b3c27bd19e4dbf94d172c308171b675a88e0df6e216cba59a078c2a923473003","observation_id":"61c9305e-0836-4499-8879-1f8d6e2c0318","resolution":{"observed_at":"2026-08-08T17:49:20.348035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.12823","last_updated":"2018-10-30T15:48:30Z","snapshot_observed_at":"2026-08-12T20:13:29.022954Z","submitted_at":"2018-10-30T15:48:30Z","title":"DeepTwist: Learning Model Compression via Occasional Weight Distortion","version":1},"cited_work":{"arxiv_id":"1810.12823","doi":null,"metadata_source":"pith","pith_arxiv_id":"1810.12823","snapshot_observed_at":"2026-08-08T17:49:20.206095Z","title":"DeepTwist: Learning Model Compression via Occasional Weight Distortion","venue":"cs.LG","work_id":"7f1a1a8d-2fd2-41e9-b35e-8c05c03fcb5d","year":2018},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.084507Z"},"links":{"cited_paper":"/paper/1810.12823","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:1ac78a44ad898895abd4252a9b39dd786ae11ebd4c750e9c1119ba63cc9bc0e7","observation_id":"1caea9bd-10e5-4a03-b927-ba1166754f8d","resolution":{"observed_at":"2026-08-08T17:49:20.211872Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.334212Z","title":"Cascaded encoders for unifying streaming and non-streaming ASR,","venue":null,"work_id":"f3a00478-3a29-475d-ae6c-1036c0f4e136","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.087835Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:6a805b6e80011e82cf9ef72bcc6e96399fad02c227cdbf920adbfed28dbfe064","observation_id":"95ab73f4-2c19-47c8-b852-3eb2037cf5c0","resolution":{"observed_at":"2026-08-08T17:49:20.338030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.323862Z","title":"Attention is all you need,","venue":null,"work_id":"ae80bac7-225b-4a0d-8eb1-3b9476cf602c","year":2017},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.090913Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:d1d0985f6fe4b94911d9e285e7af97022a6abffc8927bf2c018b6ff5061df420","observation_id":"f71628b4-fe20-44b5-bcbf-6fa3d29a08e1","resolution":{"observed_at":"2026-08-08T17:49:20.327203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.08100","last_updated":"2020-05-16T20:56:25Z","snapshot_observed_at":"2026-08-08T15:18:32.322111Z","submitted_at":"2020-05-16T20:56:25Z","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.08100","snapshot_observed_at":"2026-08-08T17:49:20.093965Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.093965Z"},"links":{"cited_paper":"/paper/2005.08100","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:7d8b15d358200a392415c066d8915b0d772c3b3045bac36276a7236a1458ac8b","observation_id":"2812bdb9-7c65-4922-b502-a03c0395e5fd","resolution":{"observed_at":"2026-08-08T17:49:20.093965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.313050Z","title":"Self-supervised learning with random- projection quantizer for speech recognition,","venue":null,"work_id":"6245fb39-d5b6-4167-8deb-54e750f1e3e2","year":2022},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.097270Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:c5842690d9eabf5820fd6a6120c2d1ec078de105a4306f13ddc4011c57ab988c","observation_id":"dea9dcf0-e6d7-4d3b-bc65-6904fb624d14","resolution":{"observed_at":"2026-08-08T17:49:20.317009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.302386Z","title":"W2v-BERT: Combining contrastive learn- ing and masked language modeling for self-supervised speech pre-training,","venue":null,"work_id":"fdbce261-dd48-4198-a886-8794f20fa265","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.100383Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:562e7d41e53c7f16a8c3a7131e4bf918bfb560fe02650c7df0bc1a85c6cb6827","observation_id":"69d4928a-e2be-4dda-b9c6-f9d6e443e208","resolution":{"observed_at":"2026-08-08T17:49:20.306332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.09586","last_updated":"2019-09-12T15:44:51Z","snapshot_observed_at":"2026-08-07T01:23:08.177613Z","submitted_at":"2019-09-12T15:44:51Z","title":"Understanding LSTM -- a tutorial into Long Short-Term Memory Recurrent Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.09586","snapshot_observed_at":"2026-08-08T17:49:20.103639Z","title":"Un- derstanding LSTM–a tutorial into long short-term memory recurrent neural networks,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.103639Z"},"links":{"cited_paper":"/paper/1909.09586","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:a8ebe751406ce1a02f26aa79f21deaed380ad4143235a6d93033ecc8f8a18293","observation_id":"deab2faf-83ff-4e99-914c-650bef3dfbdc","resolution":{"observed_at":"2026-08-08T17:49:20.103639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1211.3711","last_updated":"2012-11-14T19:25:21Z","snapshot_observed_at":"2026-08-12T19:51:35.721872Z","submitted_at":"2012-11-14T19:25:21Z","title":"Sequence Transduction with Recurrent Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1211.3711","snapshot_observed_at":"2026-08-08T17:49:20.107031Z","title":"Sequence transduction with recurrent neural networks,","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.107031Z"},"links":{"cited_paper":"/paper/1211.3711","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:99ea9a85bb65772585fb736913e86a2670e1cf65ac023c755710b44f8ed996b8","observation_id":"b9e50fc6-7dee-4fbe-a72f-e67de5025585","resolution":{"observed_at":"2026-08-08T17:49:20.107031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00390","last_updated":"2021-07-27T04:04:57Z","snapshot_observed_at":"2026-08-09T19:09:23.880235Z","submitted_at":"2021-01-02T07:24:21Z","title":"VoxPopuli: A Large-Scale Multilingual Speech Corpus for Representation Learning, Semi-Supervised Learning and Interpretation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00390","snapshot_observed_at":"2026-08-08T17:49:20.110515Z","title":"V oxPopuli: A large-scale multilin- gual speech corpus for representation learning, semi- supervised learning and interpretation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.110515Z"},"links":{"cited_paper":"/paper/2101.00390","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:88f38548864503c073d79636e94d5e7b99c5960aab33609cd0a194a2612e203e","observation_id":"8c466509-5f86-4db4-80f3-df8a3dd8b06d","resolution":{"observed_at":"2026-08-08T17:49:20.110515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.08779","last_updated":"2019-12-03T18:19:07Z","snapshot_observed_at":"2026-08-13T21:03:38.210392Z","submitted_at":"2019-04-18T17:53:38Z","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.08779","snapshot_observed_at":"2026-08-08T17:49:20.114058Z","title":"SpecAugment: A simple data augmentation method for automatic speech recognition,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.114058Z"},"links":{"cited_paper":"/paper/1904.08779","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:8609facddef710a4d5cf2a2c9710860bf6dc660b3de82395cdcfea58dd1149e2","observation_id":"039c12cd-7bc5-4e1f-8277-9b37faf27336","resolution":{"observed_at":"2026-08-08T17:49:20.114058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1508.07909","last_updated":"2016-06-10T14:45:08Z","snapshot_observed_at":"2026-08-13T18:41:35.355818Z","submitted_at":"2015-08-31T16:37:31Z","title":"Neural Machine Translation of Rare Words with Subword Units","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1508.07909","snapshot_observed_at":"2026-08-08T17:49:20.117573Z","title":"Neural machine translation of rare words with subword units,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.117573Z"},"links":{"cited_paper":"/paper/1508.07909","citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:3d382e547ff28368e0eb23ad76622d7b63654945488bddd09e4c22b0b86200a2","observation_id":"fb657d9a-3484-4c2a-aaeb-c9b978437ed2","resolution":{"observed_at":"2026-08-08T17:49:20.117573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.292202Z","title":"Comparative study of different tokenization strategies for streaming end-to-end ASR,","venue":null,"work_id":"d9bb6c2c-8ce5-4454-ba59-c475cbb5ce27","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.120910Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:c68c994eb320183c1348571efe38dd545937fb034ed0bd243578b35f5ddaba25","observation_id":"5ffa7872-b14c-48f3-becf-4ad0c4c02456","resolution":{"observed_at":"2026-08-08T17:49:20.295839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T17:49:20.281877Z","title":"Efficient knowledge distillation for rnn-transducer models,","venue":null,"work_id":"86fc0ded-5274-4b22-b2ad-0532e1eac9d7","year":2021},"citing_paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T17:49:20.124168Z"},"links":{"citing_paper":"/paper/2502.05837"},"observation_digest":"sha256:5695acc8f5fda211603ee88e26ce452c595da989fed889b77c97385e48736b90","observation_id":"993f89c4-3894-44ac-8b2a-8fc8faed0a02","resolution":{"observed_at":"2026-08-08T17:49:20.285524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.05837","last_updated":"2025-02-09T10:17:25Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-13T13:04:47.076380Z","submitted_at":"2025-02-09T10:17:25Z","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":2,"verified_fuzzy":17},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2502.05837."}