{"as_of":"2026-08-14T23:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:236510f20e2e552d4ec061672749690c9493bbee395a65a906549072c675929c","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T13:48:38.563634Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T20:37:34.442695Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-12T13:48:38.563634Z","title":"Conneau, M","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.00055","last_updated":"2024-11-24T17:02:48Z","snapshot_observed_at":"2026-08-14T09:26:53.575275Z","submitted_at":"2024-11-24T17:02:48Z","title":"High-precision medical speech recognition through synthetic data and semantic correction: UNITED-MEDASR","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T13:48:38.563634Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2412.00055"},"observation_digest":"sha256:7e9b2d31d85fb8bd488b27725ab8f165559e14c7eef99a369e1a2e7a57555ac6","observation_id":"91bdb8e0-65c8-41b7-9f00-2cf0a3ed07e7","resolution":{"observed_at":"2026-08-12T13:48:38.563634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-11T18:04:14.082404Z","title":"Preprint, arXiv:2205.12446","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.08274","last_updated":"2024-12-23T14:32:28Z","snapshot_observed_at":"2026-08-14T06:59:59.147418Z","submitted_at":"2024-12-11T10:46:21Z","title":"2M-BELEBELE: Highly Multilingual Speech and American Sign Language Comprehension Dataset","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T18:04:14.082404Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2412.08274"},"observation_digest":"sha256:b1eb0995877f68038a733204bc8bb7b11af5d78e0161f4e370e00c651dcc8b8f","observation_id":"775a4809-4b6f-4350-8c19-a6515a389042","resolution":{"observed_at":"2026-08-11T18:04:14.082404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-11T12:56:13.036672Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.13702","last_updated":"2024-12-19T17:36:38Z","snapshot_observed_at":"2026-08-14T04:38:56.692686Z","submitted_at":"2024-12-18T10:45:24Z","title":"Typhoon 2: A Family of Open Text and Multimodal Thai Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T12:56:13.036672Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2412.13702"},"observation_digest":"sha256:ab9e53fea277470343e74308c35ec10eec4bc47b75b7b5b16cc09075163b931d","observation_id":"5fa811c8-6691-417f-aa89-8646feb31940","resolution":{"observed_at":"2026-08-11T12:56:13.036672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-11T10:26:24.098338Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.16719","last_updated":"2024-12-28T17:45:12Z","snapshot_observed_at":"2026-08-14T07:28:45.983418Z","submitted_at":"2024-12-21T18:04:01Z","title":"Lillama: Large Language Models Compression via Low-Rank Feature Distillation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T10:26:24.098338Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2412.16719"},"observation_digest":"sha256:2dae9b188c23f491db11c59af347ae606116d1ab770ce67580952d9c5605f11f","observation_id":"3bfd18b1-a982-4c03-8cf9-2d6629c0ef05","resolution":{"observed_at":"2026-08-11T10:26:24.098338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-10T22:54:41.267748Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.00425","last_updated":"2024-12-31T13:03:20Z","snapshot_observed_at":"2026-08-10T22:49:09.575066Z","submitted_at":"2024-12-31T13:03:20Z","title":"Whisper Turns Stronger: Augmenting Wav2Vec 2.0 for Superior ASR in Low-Resource Languages","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T22:54:41.267748Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2501.00425"},"observation_digest":"sha256:63e7ac74e1c10267a1ed723c387b939bc2928e29dd921cb698f61f6b3b255545","observation_id":"6edc091d-4c06-4325-aef1-c506d177465e","resolution":{"observed_at":"2026-08-10T22:54:41.267748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-10T21:10:04.130795Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.06117","last_updated":"2025-08-13T20:03:11Z","snapshot_observed_at":"2026-08-14T22:00:26.207132Z","submitted_at":"2025-01-10T17:15:38Z","title":"Fleurs-SLU: A Massively Multilingual Benchmark for Spoken Language Understanding","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T21:10:04.130795Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2501.06117"},"observation_digest":"sha256:b878df81947cfa7594a91339547b751bd4a882d5c52d72eb1d03fe81308b35d2","observation_id":"6ba60897-0707-4994-9bfa-8da1648813ca","resolution":{"observed_at":"2026-08-10T21:10:04.130795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-10T20:24:34.019725Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.08696","last_updated":"2025-01-15T10:09:38Z","snapshot_observed_at":"2026-08-13T23:06:11.286332Z","submitted_at":"2025-01-15T10:09:38Z","title":"Deep Learning-Based Feature Fusion for Emotion Analysis and Suicide Risk Differentiation in Chinese Psychological Support Hotlines","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T20:24:34.019725Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2501.08696"},"observation_digest":"sha256:181b4bae6550471c7d0a5a6ca8e5aa8516f590486e22f7f4cf88fa0f1679ce81","observation_id":"ec139348-a02b-4f2a-b567-a19eaf27591d","resolution":{"observed_at":"2026-08-10T20:24:34.019725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-10T14:40:55.195830Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.15111","last_updated":"2025-01-25T07:26:37Z","snapshot_observed_at":"2026-08-13T14:25:54.211530Z","submitted_at":"2025-01-25T07:26:37Z","title":"HumanOmni: A Large Vision-Speech Language Model for Human-Centric Video Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T14:40:55.195830Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2501.15111"},"observation_digest":"sha256:d93d48c1a123906ac3f88f105c0c52c685892ec06be8bfaa5658d9558af40faa","observation_id":"5637577b-5eb4-4af3-aed5-ce92a2e3cbf4","resolution":{"observed_at":"2026-08-10T14:40:55.195830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-07T14:42:05.292113Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17841","last_updated":"2025-05-23T12:55:33Z","snapshot_observed_at":"2026-08-14T05:05:44.596019Z","submitted_at":"2025-05-23T12:55:33Z","title":"TEDI: Trustworthy and Ethical Dataset Indicators to Analyze and Compare Dataset Documentation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:42:05.292113Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2505.17841"},"observation_digest":"sha256:7d4658fdf0b0488ffc87a31449cc5ca8ec4b73067e9d6735140c7bd44ec25438","observation_id":"b4d8a558-de93-4a17-a722-b3b46e2cda41","resolution":{"observed_at":"2026-08-07T14:42:05.292113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-07T12:35:25.945196Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.24561","last_updated":"2025-05-30T13:16:08Z","snapshot_observed_at":"2026-08-12T15:49:15.982660Z","submitted_at":"2025-05-30T13:16:08Z","title":"Improving Language and Modality Transfer in Translation by Character-level Modeling","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:25.945196Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2505.24561"},"observation_digest":"sha256:bed50ad895f34bc4472212a4904137c5168a711757db25e827db01ad0c74c4cd","observation_id":"8831570c-9f43-4ccf-bf00-09910a2d9c68","resolution":{"observed_at":"2026-08-07T12:35:25.945196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-07T11:37:29.411137Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01934","last_updated":"2025-06-02T17:53:10Z","snapshot_observed_at":"2026-08-14T21:56:02.341480Z","submitted_at":"2025-06-02T17:53:10Z","title":"RoboEgo System Card: An Omnimodal Model with Native Full Duplexity","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:37:29.411137Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2506.01934"},"observation_digest":"sha256:3ffc522287cecd47e3c416fd24c8497942beb4827babbff6dd829d99aab50c1c","observation_id":"7ee4bfd6-b237-46f0-82fe-4747c7b343b2","resolution":{"observed_at":"2026-08-07T11:37:29.411137Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-06T15:41:34.664729Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15272","last_updated":"2025-07-21T06:20:27Z","snapshot_observed_at":"2026-08-08T06:16:59.918870Z","submitted_at":"2025-07-21T06:20:27Z","title":"A2TTS: TTS for Low Resource Indian Languages","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T15:41:34.664729Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2507.15272"},"observation_digest":"sha256:006f157d6a35ad0eaefea18637e8f133af40bcab23aa20bfaecc2777e29b7a5a","observation_id":"c2895bc3-dd03-4a37-8cf1-82d9040065fc","resolution":{"observed_at":"2026-08-06T15:41:34.664729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-06T15:14:29.325515Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.16875","last_updated":"2025-07-22T09:38:30Z","snapshot_observed_at":"2026-08-14T19:25:32.616653Z","submitted_at":"2025-07-22T09:38:30Z","title":"Technical report: Impact of Duration Prediction on Speaker-specific TTS for Indian Languages","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T15:14:29.325515Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2507.16875"},"observation_digest":"sha256:da8514c2e37d312c443308a13c92ae49e78dd22c46f14a57e6bf7dba6e114ff3","observation_id":"3c674f5f-94d6-4e13-8644-be859c9f748b","resolution":{"observed_at":"2026-08-06T15:14:29.325515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-05T16:54:32.879235Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2508.17494","last_updated":"2025-08-24T19:07:59Z","snapshot_observed_at":"2026-08-12T23:46:02.494743Z","submitted_at":"2025-08-24T19:07:59Z","title":"Improving French Synthetic Speech Quality via SSML Prosody Control","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T16:54:32.879235Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2508.17494"},"observation_digest":"sha256:5feac3cd15732453ebcf4e784d0eb5247213cf346204189d997d19f798bc6903","observation_id":"4b6359f2-6a4b-4e50-976b-e496ca09749d","resolution":{"observed_at":"2026-08-05T16:54:32.879235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2512.01512","last_updated":"2026-04-13T14:21:57Z","snapshot_observed_at":"2026-08-11T13:20:07.730722Z","submitted_at":"2025-12-01T10:39:12Z","title":"MCAT: Scaling Many-to-Many Speech-to-Text Translation with MLLMs to 70 Languages","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T03:15:04.685150Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2512.01512"},"observation_digest":"sha256:7cc99692427aa9deab7e0ae708cba5b5d3749299e4f7765cefd43f8dbd901108","observation_id":"de6847ad-3c01-4237-aed6-52f5a23364f3","resolution":{"observed_at":"2026-05-17T03:18:57.198237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2512.16378","last_updated":"2026-04-25T20:42:51Z","snapshot_observed_at":"2026-08-11T01:17:13.325852Z","submitted_at":"2025-12-18T10:21:14Z","title":"Hearing to Translate: The Effectiveness of Speech Modality Integration into LLMs","version":4},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-16T21:49:21.785096Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2512.16378"},"observation_digest":"sha256:8d52b56441d9f543287a6b135510b100c0eee477bbfcdba7483e4ac4b46c3778","observation_id":"f47c68f1-7d4f-4085-afb9-b4dd2c8aedce","resolution":{"observed_at":"2026-05-16T21:51:17.797958Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-03T12:01:59.543761Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.06199","last_updated":"2026-06-01T03:39:22Z","snapshot_observed_at":"2026-08-06T01:59:30.134013Z","submitted_at":"2026-01-08T07:46:03Z","title":"FastSLM: Hierarchical Temporal Abstraction for Efficient Long-Form Speech Adaptation","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T12:01:59.543761Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2601.06199"},"observation_digest":"sha256:b0484b42cdd6d2f6d91ab5daf42e0a142926118a08895053767ad02a16727e03","observation_id":"98e7d417-f76a-4920-b990-0765c66a7a23","resolution":{"observed_at":"2026-08-03T12:01:59.543761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-15T00:03:31.986628Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.09725","last_updated":"2026-07-09T13:30:49Z","snapshot_observed_at":"2026-08-14T23:25:21.924666Z","submitted_at":"2026-03-10T14:32:12Z","title":"A Semi-spontaneous Dutch Speech Dataset for Speech Enhancement and Speech Recognition","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-07-15T00:03:31.986628Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2603.09725"},"observation_digest":"sha256:3361d79d42681a7bba8673eab0c5d2988389e97939c0046023da83774cc08003","observation_id":"a359ee39-045e-4391-90ac-d2c9dfef6e8f","resolution":{"observed_at":"2026-07-15T00:03:31.986628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2604.04598","last_updated":"2026-04-06T11:23:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T11:23:42Z","title":"Benchmarking Multilingual Speech Models on Pashto: Zero-Shot ASR, Script Failure, and Cross-Domain Evaluation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T19:44:30.762851Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2604.04598"},"observation_digest":"sha256:2401794c97ccfc771eb928b7ab5c75a0bfe9b8430e5755b6973be39bb1c4d09a","observation_id":"3c5020b0-ccdd-4bc9-ae71-87a0209c184a","resolution":{"observed_at":"2026-05-10T22:35:48.735609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2604.10736","last_updated":"2026-04-16T21:22:01Z","snapshot_observed_at":"2026-07-06T22:59:18.756089Z","submitted_at":"2026-04-12T17:17:54Z","title":"BlasBench: An Open Benchmark for Irish Speech Recognition","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T15:53:54.092426Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2604.10736"},"observation_digest":"sha256:12dee420ea6f24fd7e33d271841d23c029def19901b05db5409c7e3088a8bd22","observation_id":"f2e50f6c-090e-44e6-a174-a2753e2fba7c","resolution":{"observed_at":"2026-05-11T09:41:01.686680Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2604.19221","last_updated":"2026-04-30T07:45:08Z","snapshot_observed_at":"2026-08-11T09:39:20.137545Z","submitted_at":"2026-04-21T08:24:55Z","title":"UAF: A Unified Audio Front-end LLM for Full-Duplex Speech Interaction","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T03:05:48.624864Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2604.19221"},"observation_digest":"sha256:fb5f5b8cf02b22cc3db70ee0cd7c700ed0878951258b2ef943d61fcc6ac96fca","observation_id":"0b670554-ff01-4049-983c-a55df6dee226","resolution":{"observed_at":"2026-05-11T12:46:02.907523Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2604.19565","last_updated":"2026-04-21T15:18:10Z","snapshot_observed_at":"2026-08-11T12:12:54.194536Z","submitted_at":"2026-04-21T15:18:10Z","title":"Detecting Hallucinations in SpeechLLMs at Inference Time Using Attention Maps","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-10T01:56:44.270045Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2604.19565"},"observation_digest":"sha256:e1728a733260fdd44b3d34498196cfcad8200f8d6419d27c95029f9907d4e237","observation_id":"f41d2b04-5582-4c5f-baab-9aa025c9e7c3","resolution":{"observed_at":"2026-05-11T13:21:04.999089Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2605.13087","last_updated":"2026-06-29T10:37:57Z","snapshot_observed_at":"2026-08-11T03:07:50.845711Z","submitted_at":"2026-05-13T06:55:55Z","title":"Vividh-ASR: A Complexity-Tiered Benchmark and Optimization Dynamics for Robust Indic Speech Recognition","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-14T19:37:21.180532Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.13087"},"observation_digest":"sha256:a38c312cbffe687bfab76c94d679b08f1977c2507fa1c3061bf5ca072e2e41ad","observation_id":"7675bf7d-1c01-4393-920c-9b7726f32a76","resolution":{"observed_at":"2026-05-14T19:37:51.725042Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2605.13087","last_updated":"2026-06-29T10:37:57Z","snapshot_observed_at":"2026-08-11T03:07:50.845711Z","submitted_at":"2026-05-13T06:55:55Z","title":"Vividh-ASR: A Complexity-Tiered Benchmark and Optimization Dynamics for Robust Indic Speech Recognition","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-30T21:54:06.330240Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.13087"},"observation_digest":"sha256:f91032cbe92aabec7e0e2cfd722209d75fef3636b774d2ef33876e2abdf13d14","observation_id":"f68da92d-ff51-4373-9c11-d06c92d07b0f","resolution":{"observed_at":"2026-06-30T21:55:05.379459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2605.23463","last_updated":"2026-05-22T10:24:50Z","snapshot_observed_at":"2026-08-12T19:51:32.587082Z","submitted_at":"2026-05-22T10:24:50Z","title":"StepAudio 2.5 Technical Report","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-25T02:52:22.610397Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.23463"},"observation_digest":"sha256:09f9cbbc13dedc030c0f66870c8113a228157d5f79d41ae0d7766f3ff6192d32","observation_id":"49a49c25-9c27-466e-ae3e-3b83d2481bb9","resolution":{"observed_at":"2026-05-25T02:55:16.447086Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-13T08:22:44.347941Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2605.23912","last_updated":"2026-04-08T23:43:46Z","snapshot_observed_at":"2026-08-14T10:46:34.269607Z","submitted_at":"2026-04-08T23:43:46Z","title":"Raon-Speech Technical Report","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-13T08:22:44.347941Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.23912"},"observation_digest":"sha256:256e78ddd1ad34a5ac1d4f90c5cde442245ff5daf01ff322b13ea17d33838e76","observation_id":"ba3153f9-1fff-43e4-b91e-ca22e7eeda19","resolution":{"observed_at":"2026-07-13T08:22:44.347941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2605.25596","last_updated":"2026-08-02T16:42:39Z","snapshot_observed_at":"2026-08-10T17:35:16.812149Z","submitted_at":"2026-05-25T08:47:33Z","title":"Multilingual Phonological Feature Recognition with Self-Supervised Speech Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-29T21:35:03.961889Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.25596"},"observation_digest":"sha256:1fb4685a0f540e9c76fae76e7a44193255788e050532e3f276ebed718877f1dc","observation_id":"499d8a2b-4a8d-4b5e-8845-cc7f2a431d92","resolution":{"observed_at":"2026-06-29T21:43:59.729604Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-04T05:02:46.329258Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2605.25596","last_updated":"2026-08-02T16:42:39Z","snapshot_observed_at":"2026-08-10T17:35:16.812149Z","submitted_at":"2026-05-25T08:47:33Z","title":"Multilingual Phonological Feature Recognition with Self-Supervised Speech Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T05:02:46.329258Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.25596"},"observation_digest":"sha256:79442300aebbc6a3398232ea32419b8fb29f18d1bc2a2f344ea7d799abf64b34","observation_id":"aede9838-eb4c-4e07-8eda-5847a892c127","resolution":{"observed_at":"2026-08-04T05:02:46.329258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2605.28211","last_updated":"2026-05-27T09:30:36Z","snapshot_observed_at":"2026-08-13T01:51:17.024473Z","submitted_at":"2026-05-27T09:30:36Z","title":"When Helpful Context Leaks: Privacy Risks in Domain-Adapted ASR","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T12:35:08.845406Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2605.28211"},"observation_digest":"sha256:03eb3e7cfc9884d8dc060aa6171f96be4a31545e69888b78e650a8a2b3f2c82c","observation_id":"336777c9-5e30-488c-a644-23d88ad7bc59","resolution":{"observed_at":"2026-06-29T12:43:25.969323Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2606.09335","last_updated":"2026-06-08T11:03:25Z","snapshot_observed_at":"2026-08-07T05:59:53.128999Z","submitted_at":"2026-06-08T11:03:25Z","title":"Factors affecting ASR performance: A study using state of the art ASR models in Indic Languages","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T15:10:03.282212Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2606.09335"},"observation_digest":"sha256:9419f73b612070a7b29ef6ce8ffd8a378c2e145da33ac60cf2083cd764c52e31","observation_id":"88655144-e28f-404b-8213-18b620ead68a","resolution":{"observed_at":"2026-07-03T03:37:35.500051Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-11T09:35:44.381743Z","title":"FLEURS: Few- shot learning evaluation of universal representations of speech,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.05051","last_updated":"2026-07-06T13:25:33Z","snapshot_observed_at":"2026-08-13T01:49:56.538580Z","submitted_at":"2026-07-06T13:25:33Z","title":"Listen, Think, Transcribe: Continuous Latent Test-Time Scaling for ASR","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-11T09:35:44.381743Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.05051"},"observation_digest":"sha256:093a8266ac9e3933ab93856173277160a863128648e96f945dfa326affd0c88e","observation_id":"bab88ca9-b577-4721-8e2b-fbe6da7cfdfa","resolution":{"observed_at":"2026-07-11T09:35:44.381743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":"2205.12446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-07-10T20:37:34.442695Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech","venue":"cs.CL","work_id":"591017b1-e3a5-4f06-9d81-998d68d87365","year":2022},"citing_paper":{"arxiv_id":"2607.06827","last_updated":"2026-07-07T21:45:04Z","snapshot_observed_at":"2026-07-11T23:18:39.332726Z","submitted_at":"2026-07-07T21:45:04Z","title":"Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-10T20:30:11.127007Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.06827"},"observation_digest":"sha256:4b79a233f43e93fa4dff47430d4abd2461c42bb87969efb3b89382f4a3e8aee2","observation_id":"fb2b9101-cb2b-4880-9cb4-6c6957330305","resolution":{"observed_at":"2026-07-10T20:37:34.444265Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-01T17:45:41.785420Z","title":"arXiv preprint arXiv:2205.12446 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17544","last_updated":"2026-07-20T04:42:07Z","snapshot_observed_at":"2026-08-13T06:15:37.388922Z","submitted_at":"2026-07-20T04:42:07Z","title":"X-Translator: A Real-Time Multilingual Speaker-Aware Speech-to-Speech Translation System","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-01T17:45:41.785420Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.17544"},"observation_digest":"sha256:d02067f889b5ffafbc70aca3229ab358c251b5c530e80c1c3e357dfe30742f74","observation_id":"b9f8c728-53b6-43b0-8ad9-6951928f0dd8","resolution":{"observed_at":"2026-08-01T17:45:41.785420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-01T05:50:31.979267Z","title":"Fleurs: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.22100","last_updated":"2026-07-24T08:52:44Z","snapshot_observed_at":"2026-08-13T04:53:52.000451Z","submitted_at":"2026-07-24T08:52:44Z","title":"MEUSLI: a Multilingual Projector for LLM-based ASR and Beyond","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-01T05:50:31.979267Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.22100"},"observation_digest":"sha256:71423f53d3e78f2b0b172731e5678d3f9aeeb901012ca5a5ceb40f255a506041","observation_id":"c1eec3a5-1a28-4ebe-9e99-2a11e7f5e83a","resolution":{"observed_at":"2026-08-01T05:50:31.979267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-03T10:13:54.959703Z","title":"arXiv preprint arXiv:2205.12446 (2022)","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29279","last_updated":"2026-07-31T10:50:14Z","snapshot_observed_at":"2026-08-06T04:02:53.771607Z","submitted_at":"2026-07-31T10:50:14Z","title":"ParaASR: Multi-Token Prediction for Fast and Long-Context LLM-Based Speech Recognition","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T10:13:54.959703Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2607.29279"},"observation_digest":"sha256:260d0912b1352c4d6fa957675e72f8ff9ce1a51ee22b6a0c825446cb5b3f64b3","observation_id":"0a00891e-d3f5-4e73-b27d-8578017b231d","resolution":{"observed_at":"2026-08-03T10:13:54.959703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.12446","snapshot_observed_at":"2026-08-12T11:39:37.418590Z","title":"arXiv preprint arXiv:2205.12446 , url =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11036","last_updated":"2026-08-11T15:12:42Z","snapshot_observed_at":"2026-08-14T23:12:28.540559Z","submitted_at":"2026-08-11T15:12:42Z","title":"myMediWhisper: Construction of Burmese Medical Speech Corpus and Whisper Fine-Tuning for Clinical Dialogue ASR","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-12T11:39:37.418590Z"},"links":{"cited_paper":"/paper/2205.12446","citing_paper":"/paper/2608.11036"},"observation_digest":"sha256:31b8b79ff3d3379ef04fd330f4b4586a70c98163593529ab9f323d2677d8b2fb","observation_id":"be7e8183-8dd7-4496-8cf1-b13891e985a4","resolution":{"observed_at":"2026-08-12T11:39:37.418590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2205.12446/citation-record","integrity":"/paper/2205.12446/integrity","json":"/paper/2205.12446/citation-record.json","paper":"/paper/2205.12446"},"outbound":[],"paper":{"arxiv_id":"2205.12446","last_updated":"2022-05-25T02:29:03Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T21:07:39.743665Z","submitted_at":"2022-05-25T02:29:03Z","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2205.12446."}