{"as_of":"2026-08-21T04:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d19e41a5aa48ca909c79498c5bd15dd691ab1b0e50e028e956ef7d9d65e921e6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":34,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T00:47:52.584754Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":25,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-12T14:33:49.954272Z","title":"Hubert: Self-supervised speech representation learning by masked reconstruction,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.15082","last_updated":"2024-11-22T17:18:08Z","snapshot_observed_at":"2026-08-17T12:17:45.228042Z","submitted_at":"2024-11-22T17:18:08Z","title":"Towards Speaker Identification with Minimal Dataset and Constrained Resources using 1D-Convolution Neural Network","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T14:33:49.954272Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2411.15082"},"observation_digest":"sha256:75da91a9467868e74e714f74662bbc6554596380d1ef5002dba6019e918fe6d7","observation_id":"063db37c-f8db-4c10-af47-3c256df3285b","resolution":{"observed_at":"2026-08-12T14:33:49.954272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-10T20:27:22.494754Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.16344","last_updated":"2025-05-31T16:37:32Z","snapshot_observed_at":"2026-08-16T17:04:21.935192Z","submitted_at":"2025-01-15T06:30:17Z","title":"WhiSPA: Semantically and Psychologically Aligned Whisper with Self-Supervised Contrastive and Student-Teacher Learning","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T20:27:22.494754Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2501.16344"},"observation_digest":"sha256:c80a31b51ca3d406a9c9f906fed2a6d91ee8d8090422d769c0c0cdc17a31b907","observation_id":"dc48ff5f-8d49-4f63-8941-bf4181efd1bc","resolution":{"observed_at":"2026-08-10T20:27:22.494754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-16T00:47:52.584754Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.02707","last_updated":"2025-05-05T15:05:01Z","snapshot_observed_at":"2026-08-18T17:55:59.387301Z","submitted_at":"2025-05-05T15:05:01Z","title":"Voila: Voice-Language Foundation Models for Real-Time Autonomous Interaction and Voice Role-Play","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-16T00:47:52.584754Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2505.02707"},"observation_digest":"sha256:9f519136ffc076a8fc636cf8d88f7e11b0dce0313e49b128f7dda5c8279828df","observation_id":"dd19769d-df76-4e6a-b750-8aebb223ab4f","resolution":{"observed_at":"2026-08-16T00:47:52.584754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-15T21:03:18.019220Z","title":"arXiv:2106.07447","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10975","last_updated":"2026-05-27T20:17:05Z","snapshot_observed_at":"2026-08-19T08:21:18.467991Z","submitted_at":"2025-05-16T08:21:59Z","title":"Survey of End-to-End Multi-Speaker Automatic Speech Recognition for Monaural Audio","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T21:03:18.019220Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2505.10975"},"observation_digest":"sha256:cbbffbd01f95deb25f6028b28bbc0779a3be1f9711abda4e3b8c30088a65596d","observation_id":"43644128-e3fd-42f2-83bc-2ddc70e9d148","resolution":{"observed_at":"2026-08-15T21:03:18.019220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T14:58:57.191969Z","title":"Preprint, arXiv:2106.07447","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16691","last_updated":"2025-05-23T05:07:17Z","snapshot_observed_at":"2026-08-13T05:00:06.234822Z","submitted_at":"2025-05-22T13:57:02Z","title":"EZ-VC: Easy Zero-shot Any-to-Any Voice Conversion","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T14:58:57.191969Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2505.16691"},"observation_digest":"sha256:e31c09423040623bb80deaf89d6c0ffc12c23037fdc2be8834a66fa513555e43","observation_id":"3efc365d-704a-4714-8d27-54e70b726133","resolution":{"observed_at":"2026-08-07T14:58:57.191969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T05:20:04.453086Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.08372","last_updated":"2025-06-10T02:37:42Z","snapshot_observed_at":"2026-08-19T08:36:50.636957Z","submitted_at":"2025-06-10T02:37:42Z","title":"Multimodal Zero-Shot Framework for Deepfake Hate Speech Detection in Low-Resource Languages","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:20:04.453086Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.08372"},"observation_digest":"sha256:e0c59d9ae1b42fb57b744f252ec19152950333fc65c4a12694a7c564f628a99a","observation_id":"c86a60e8-64fa-429d-b203-491345f93955","resolution":{"observed_at":"2026-08-07T05:20:04.453086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-07T05:26:36.292242Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.11119","last_updated":"2025-06-09T17:52:31Z","snapshot_observed_at":"2026-08-07T05:18:46.575540Z","submitted_at":"2025-06-09T17:52:31Z","title":"Benchmarking Foundation Speech and Language Models for Alzheimer's Disease and Related Dementia Detection from Spontaneous Speech","version":1},"reference_index":3460,"source":"pdf_text","source_observed_at":"2026-08-07T05:26:36.292242Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.11119"},"observation_digest":"sha256:01c31d953089940b1f75b0655cc326529009253c7b0d16d88e33e52aca2e28bc","observation_id":"b907b8c8-5dd0-4009-a434-2028a84c0a45","resolution":{"observed_at":"2026-08-07T05:26:36.292242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T23:41:49.569844Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.17351","last_updated":"2025-06-20T01:28:43Z","snapshot_observed_at":"2026-08-14T02:58:18.356312Z","submitted_at":"2025-06-20T01:28:43Z","title":"Zero-Shot Cognitive Impairment Detection from Speech Using AudioLLM","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:41:49.569844Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.17351"},"observation_digest":"sha256:e25ffa2a638c543afa14b4968a8d9cbd872599a922a3028518660ae61ccb87eb","observation_id":"2757860d-2e1a-4049-800c-f969eeb485a8","resolution":{"observed_at":"2026-08-06T23:41:49.569844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T21:51:35.040172Z","title":"Shengpeng Ji, Yifu Chen, Minghui Fang, Jialong Zuo, Jingyu Lu, Hanting Wang, Ziyue Jiang, Long Zhou, Shujie Liu, Xize Cheng, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23325","last_updated":"2025-07-09T17:40:35Z","snapshot_observed_at":"2026-08-17T17:41:09.789585Z","submitted_at":"2025-06-29T16:51:50Z","title":"XY-Tokenizer: Mitigating the Semantic-Acoustic Conflict in Low-Bitrate Speech Codecs","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:51:35.040172Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.23325"},"observation_digest":"sha256:f2716d704a524669c207d4bbaa4023229bce95ba650cdb5693a18e8dcdd40d41","observation_id":"762248e0-df10-4ea3-ac8f-a499ded69049","resolution":{"observed_at":"2026-08-06T21:51:35.040172Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T21:36:11.289042Z","title":"Hubert: Self- supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.23869","last_updated":"2025-06-30T14:00:14Z","snapshot_observed_at":"2026-08-15T16:41:27.985521Z","submitted_at":"2025-06-30T14:00:14Z","title":"Scaling Self-Supervised Representation Learning for Symbolic Piano Performance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:36:11.289042Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2506.23869"},"observation_digest":"sha256:2f1a2e301cd44a80559e6863a599d21598c55bf0460494775f0a4d1cb454fb18","observation_id":"ca1dcac3-5898-4a56-8ae9-a51f4ae5a008","resolution":{"observed_at":"2026-08-06T21:36:11.289042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T22:56:46.224028Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02915","last_updated":"2025-06-25T08:38:27Z","snapshot_observed_at":"2026-08-19T14:49:46.058264Z","submitted_at":"2025-06-25T08:38:27Z","title":"Audio-JEPA: Joint-Embedding Predictive Architecture for Audio Representation Learning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:56:46.224028Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.02915"},"observation_digest":"sha256:fecedd01f6d631369ba273e88ee40a215ddb2790c6e295648f347415a51c33bb","observation_id":"2f1a916e-e772-4f3d-8282-a493595b354f","resolution":{"observed_at":"2026-08-06T22:56:46.224028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T19:50:22.355340Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04554","last_updated":"2025-07-08T12:27:54Z","snapshot_observed_at":"2026-08-14T02:58:22.557056Z","submitted_at":"2025-07-06T22:11:22Z","title":"Self-supervised learning of speech representations with Dutch archival data","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:50:22.355340Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.04554"},"observation_digest":"sha256:2760d869d34c071203cf3db57f2d2e119a1059b8daefebada06bfdbdcddfb5a3","observation_id":"ff90399d-1651-42e3-97fb-5e6f8bcc7037","resolution":{"observed_at":"2026-08-06T19:50:22.355340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T15:31:18.775855Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.15641","last_updated":"2025-07-21T14:03:08Z","snapshot_observed_at":"2026-08-16T18:45:00.132141Z","submitted_at":"2025-07-21T14:03:08Z","title":"Leveraging Context for Multimodal Fallacy Classification in Political Debates","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T15:31:18.775855Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.15641"},"observation_digest":"sha256:735f9aa8cd82d30668cd54886f252d21a1c6c591d7f1c79be85290f3f760baa6","observation_id":"e26faa88-32ee-45c7-8e30-07a804b5bb7d","resolution":{"observed_at":"2026-08-06T15:31:18.775855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-08-13T23:38:52.637908Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:6a9f626d48680f7202ed09d2dfba96e2ab0920af599db86d8b4b3fefd5bc865a","observation_id":"986ec86d-5bf2-489e-a7fd-de4c8516c67d","resolution":{"observed_at":"2026-05-16T05:59:51.132301Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-06T14:49:52.253608Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.17799","last_updated":"2025-07-23T16:11:44Z","snapshot_observed_at":"2026-08-13T23:40:52.624764Z","submitted_at":"2025-07-23T16:11:44Z","title":"A Concept-based approach to Voice Disorder Detection","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T14:49:52.253608Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2507.17799"},"observation_digest":"sha256:505d1bb9a600e9aa554786dfd7be92706f74de8ad69399fbb0f1dfa842720418","observation_id":"b7d3a70e-6d86-4d32-b58b-f91ce0aac5bf","resolution":{"observed_at":"2026-08-06T14:49:52.253608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T17:33:58.485256Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.16188","last_updated":"2025-08-27T19:49:56Z","snapshot_observed_at":"2026-08-16T13:50:39.724461Z","submitted_at":"2025-08-22T08:08:45Z","title":"Seeing is Believing: Emotion-Aware Audio-Visual Language Modeling for Expressive Speech Generation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T17:33:58.485256Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2508.16188"},"observation_digest":"sha256:f194e2bcf27cbd91270b88922c3b76e23f947b12696b8b0fe6c1f5077d69f2b7","observation_id":"d20f6562-f71c-4c16-8d5f-bdf6236aaf4f","resolution":{"observed_at":"2026-08-05T17:33:58.485256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T13:36:04.730722Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00503","last_updated":"2025-08-30T13:50:58Z","snapshot_observed_at":"2026-08-21T02:55:34.176877Z","submitted_at":"2025-08-30T13:50:58Z","title":"Entropy-based Coarse and Compressed Semantic Speech Representation Learning","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T13:36:04.730722Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2509.00503"},"observation_digest":"sha256:28dd3a3081f9a71a8808de80ed91247ed2a3b37d48b1e0284b4637dafdf974d9","observation_id":"c9954816-dbc8-43e3-9707-98f3ef9225ec","resolution":{"observed_at":"2026-08-05T13:36:04.730722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-04T13:15:22.832912Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2510.01157","last_updated":"2026-06-09T20:25:55Z","snapshot_observed_at":"2026-08-13T08:13:31.076744Z","submitted_at":"2025-10-01T17:45:04Z","title":"Where Do Backdoors Live? A Component-Level Analysis of Backdoor Propagation in Speech Language Models","version":4},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T13:15:22.832912Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2510.01157"},"observation_digest":"sha256:911b4d4e190082e6e3703cea00699d211b23bbe69cbf15480cafa967544b292e","observation_id":"a0f2b216-8a76-4676-9c08-b3ca6222aa23","resolution":{"observed_at":"2026-08-04T13:15:22.832912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2603.12221","last_updated":"2026-04-13T08:30:07Z","snapshot_observed_at":"2026-07-06T22:48:53.198077Z","submitted_at":"2026-03-12T17:45:12Z","title":"A Two-Stage Dual-Modality Model for Facial Emotional Expression Recognition","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T11:35:38.330061Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2603.12221"},"observation_digest":"sha256:0ce60b74aa83c351d0a6067e3a81b76ab54b1c1e38bac75cd9906fb38a0650b8","observation_id":"14a60961-3961-41d0-a5f4-b6cea4c81c02","resolution":{"observed_at":"2026-05-15T11:39:59.378823Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-13T17:37:12.659819Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.26292","last_updated":"2026-06-15T19:50:12Z","snapshot_observed_at":"2026-08-14T02:58:21.710353Z","submitted_at":"2026-03-27T11:03:08Z","title":"findsylls: A Language-Agnostic Toolkit for Syllable-Level Speech Tokenization and Embedding","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-13T17:37:12.659819Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2603.26292"},"observation_digest":"sha256:9cc4f88b94cf2ad5ffcda9650cd15b576711a0db8fb8827fbf23c9d174e7ee22","observation_id":"baacf565-a834-462f-bf1b-b9e2c8474f3a","resolution":{"observed_at":"2026-07-13T17:37:12.659819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2604.08562","last_updated":"2026-03-17T16:07:15Z","snapshot_observed_at":"2026-08-15T03:13:18.639877Z","submitted_at":"2026-03-17T16:07:15Z","title":"Neural networks for Text-to-Speech evaluation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T09:46:26.884551Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2604.08562"},"observation_digest":"sha256:6af6e8cf421c4125861fe45296ff45adcbdb6e474e9642b0e85c36e0144a4f71","observation_id":"f7728530-8560-4a6b-9fcb-4bc70517722d","resolution":{"observed_at":"2026-05-15T09:49:54.655165Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2605.09152","last_updated":"2026-05-09T20:30:15Z","snapshot_observed_at":"2026-08-15T05:50:55.866969Z","submitted_at":"2026-05-09T20:30:15Z","title":"Meow-Omni 1: A Multimodal Large Language Model for Feline Ethology","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T04:02:23.847187Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2605.09152"},"observation_digest":"sha256:34dda9826388934874f7bd657d7d8a64d52d8b9b1c2bd00ce442e91e2db26a0f","observation_id":"b1c58a5a-15e7-47fa-849e-a00281d74a84","resolution":{"observed_at":"2026-05-12T06:41:43.777162Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2605.19224","last_updated":"2026-05-19T00:49:51Z","snapshot_observed_at":"2026-08-16T11:23:37.818245Z","submitted_at":"2026-05-19T00:49:51Z","title":"Fine-tuning language encoding models on slow fMRI improves prediction for fast ECoG","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T06:48:52.078640Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2605.19224"},"observation_digest":"sha256:5fca23cc4e6e218fb70f387ab65ef9420a8972b941a5f7c7e6b312f370d6feff","observation_id":"590cc051-77a2-40c2-9753-b2bdf1d71bc4","resolution":{"observed_at":"2026-05-20T06:53:06.014414Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.06357","last_updated":"2026-06-04T16:25:07Z","snapshot_observed_at":"2026-08-12T03:15:42.764171Z","submitted_at":"2026-06-04T16:25:07Z","title":"F3-Tokenizer: Taming Audio Autoencoder Latents for Understanding and Generation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:36:27.369551Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.06357"},"observation_digest":"sha256:88eb922db70aaaaffbb93386815c0fd23d949e2203cc3d9fda19b18d31f7a733","observation_id":"27588fde-9bd5-4255-a614-0f6e73963047","resolution":{"observed_at":"2026-07-02T15:47:06.020988Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.11542","last_updated":"2026-06-10T01:07:32Z","snapshot_observed_at":"2026-08-13T11:50:56.726639Z","submitted_at":"2026-06-10T01:07:32Z","title":"Pretrained self-supervised speech models can recognize unseen consonants","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T10:14:47.932613Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.11542"},"observation_digest":"sha256:345b9e896f93f04c3afa7b0df2ca00d75c0d023e2f134da8ce45a47b5611fb39","observation_id":"fcef0e10-8117-47e2-959c-c7bf9d3907e5","resolution":{"observed_at":"2026-07-03T10:07:56.083448Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.19910","last_updated":"2026-06-23T10:40:32Z","snapshot_observed_at":"2026-08-17T12:31:53.029477Z","submitted_at":"2026-06-18T08:04:16Z","title":"Light-weight Pronunciation Assessment via Discrete Speech Token Surprisal","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-26T17:37:25.043607Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.19910"},"observation_digest":"sha256:131d056e424e5a80c0ed0ca71785075a007c65a9d032469971f4057732d41793","observation_id":"46238d6e-7817-465c-bc14-d6371a8c32b9","resolution":{"observed_at":"2026-07-04T03:49:30.223989Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.22473","last_updated":"2026-06-21T12:33:44Z","snapshot_observed_at":"2026-08-14T14:26:37.762631Z","submitted_at":"2026-06-21T12:33:44Z","title":"Interleaved Speech Language Models Latently Work In Text","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T10:41:19.777779Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.22473"},"observation_digest":"sha256:a34396318df71e6d640f2a8e22a8aff3aff1319c2a5bb7da614198032b2f05f8","observation_id":"4cba52db-200a-4e08-9c0a-e31d1b391a8c","resolution":{"observed_at":"2026-07-04T08:59:42.891310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.24910","last_updated":"2026-06-19T08:07:13Z","snapshot_observed_at":"2026-08-13T09:40:37.751064Z","submitted_at":"2026-06-19T08:07:13Z","title":"End-to-End Voice Intent Recognition for Spontaneous Human-Drone Interaction with Naive Users","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-26T13:30:12.101045Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.24910"},"observation_digest":"sha256:ce430f75015469e5d79cef6f47cf0cb215ae27c643d04a265f7177ab6d46bcf9","observation_id":"12d732ad-24e4-49be-9020-97b89026ede6","resolution":{"observed_at":"2026-07-04T07:29:38.221570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2606.27206","last_updated":"2026-06-25T16:02:14Z","snapshot_observed_at":"2026-08-12T13:06:59.794611Z","submitted_at":"2026-06-25T16:02:14Z","title":"Syntactic Belief Update as the Driver of Garden Path Processing Difficulty","version":1},"reference_index":288,"source":"arxiv_source","source_observed_at":"2026-06-26T04:38:01.183423Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2606.27206"},"observation_digest":"sha256:df2cae12451c88109126b5334f0cee3ee3f5341b546e8250f0b61cf398b42168","observation_id":"05e6240c-cfc1-42e9-b23f-5ac9f84ef1b8","resolution":{"observed_at":"2026-06-26T04:38:58.619776Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-13T01:58:25.310025Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:f58b4f2b9374041e1a5b025d885f7d1fa89bbd6dc99bed77db4511426f6cdd9d","observation_id":"28de5576-2bb0-49ca-9446-7276fb253d47","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-12T06:06:46.343211Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02920","last_updated":"2026-07-03T03:24:38Z","snapshot_observed_at":"2026-08-19T17:04:02.832931Z","submitted_at":"2026-07-03T03:24:38Z","title":"Layer-wise Cross-Lingual Depression Detection from Speech: Analysis with Contrastive Alignment","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T06:06:46.343211Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.02920"},"observation_digest":"sha256:4dc598e09508884a161b9ac92250a5d02b4916e30a99fd51831c37198cfacde5","observation_id":"2e838a15-a1a9-403d-9d33-4fc1b0085c1f","resolution":{"observed_at":"2026-07-12T06:06:46.343211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":"2106.07447","doi":"10.48550/arxiv.2106.07447","metadata_source":"pith","pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"doi: 10.1088/ 0954-898X_15_2_002","venue":"cs.CL","work_id":"9caaa4c6-8050-4f22-ba70-f6ca40365168","year":2021},"citing_paper":{"arxiv_id":"2607.06875","last_updated":"2026-07-08T00:17:20Z","snapshot_observed_at":"2026-08-20T01:48:53.722098Z","submitted_at":"2026-07-08T00:17:20Z","title":"Video2Reaction: Mapping Video to Audience Reaction Distribution in the Wild","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-09T23:51:55.422232Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.06875"},"observation_digest":"sha256:daa508e39921c62dfa1198429284a4567a0fc47a5fd76895d9e3a74aa67c351c","observation_id":"3df35c21-9c57-4186-97af-02c2123f6fe0","resolution":{"observed_at":"2026-07-09T23:56:38.448201Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-01T07:10:39.830211Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.21540","last_updated":"2026-07-24T18:05:20Z","snapshot_observed_at":"2026-08-18T11:39:45.613825Z","submitted_at":"2026-07-23T17:25:08Z","title":"DONDO: Open w2v-BERT Speech-Recognition Base Models for African Languages","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T07:10:39.830211Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.21540"},"observation_digest":"sha256:62e895d0df4f806339df05e3452ff7cb42b953681ab336fcce4a259b41dc1043","observation_id":"0c6ac7ef-ce87-48c6-acb2-7e2bfde25cf9","resolution":{"observed_at":"2026-08-01T07:10:39.830211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-08-14T04:32:47.948785Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.08667","last_updated":"2026-08-09T12:21:54Z","snapshot_observed_at":"2026-08-14T09:18:48.936467Z","submitted_at":"2026-08-09T12:21:54Z","title":"A Unifying Perspective on Audio Generative Modeling: Latent Representations and Modeling Strategies","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-14T04:32:47.948785Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2608.08667"},"observation_digest":"sha256:05a6b391216122090336f4e927ca4706a302260bf0affea76577543bd31b8612","observation_id":"d49c1e6b-732d-4b39-9099-322895f7ec7b","resolution":{"observed_at":"2026-08-14T04:32:47.948785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2106.07447/citation-record","integrity":"/paper/2106.07447/integrity","json":"/paper/2106.07447/citation-record.json","paper":"/paper/2106.07447"},"outbound":[],"paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T11:07:54.315530Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 34 inbound Pith citation observations for arXiv:2106.07447."}