{"as_of":"2026-08-09T18:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6ac466bc227ee30ff18029fc3a098dca2d088b39949b862d1e44efc3f688d1cb","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:56:29.985902Z","state":"measured"},{"denominator":46,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":46,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:56:25.830154Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.20564","snapshot_observed_at":"2026-08-07T13:56:25.830154Z","title":"Figure 1 illustrates our pipeline for creating the NaijaV oices dataset","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:25.830154Z"},"links":{"cited_paper":"/paper/2505.20564","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:3d551e92d4d9e7515ce8df320831fed4438fa7ab3b151dfe79a5e1bdf6a46bd6","observation_id":"3d6feb57-e0c6-4767-b7e1-11026b590d00","resolution":{"observed_at":"2026-08-07T13:56:25.830154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"cited_work":{"arxiv_id":"2505.20564","doi":"10.48550/arxiv.2505.20564","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20564","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The naijavoices dataset: Cultivating large-scale, high-quality, culturally-rich speech data for african languages","venue":"ArXiv.org","work_id":"656f006a-4768-4773-a43e-2828cafe975f","year":2025},"citing_paper":{"arxiv_id":"2605.01597","last_updated":"2026-05-02T20:11:12Z","snapshot_observed_at":"2026-08-03T00:52:48.762187Z","submitted_at":"2026-05-02T20:11:12Z","title":"Toward Fair Speech Technologies: A Comprehensive Survey of Bias and Fairness in Speech AI","version":1},"reference_index":166,"source":"pdf_text","source_observed_at":"2026-05-08T19:27:18.774649Z"},"links":{"cited_paper":"/paper/2505.20564","citing_paper":"/paper/2605.01597"},"observation_digest":"sha256:caf875826d3fc7a2f9202179573b992d0b2fed2acc17b6afcf89ac7d4df64abd","observation_id":"6771a249-08d4-4e15-a0d4-0f90b1c86d08","resolution":{"observed_at":"2026-05-09T05:50:28.342515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"cited_work":{"arxiv_id":"2505.20564","doi":"10.48550/arxiv.2505.20564","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20564","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The naijavoices dataset: Cultivating large-scale, high-quality, culturally-rich speech data for african languages","venue":"ArXiv.org","work_id":"656f006a-4768-4773-a43e-2828cafe975f","year":2025},"citing_paper":{"arxiv_id":"2606.11219","last_updated":"2026-05-11T20:27:40Z","snapshot_observed_at":"2026-08-05T20:46:41.277498Z","submitted_at":"2026-05-11T20:27:40Z","title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","version":1},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-06-30T22:11:44.891731Z"},"links":{"cited_paper":"/paper/2505.20564","citing_paper":"/paper/2606.11219"},"observation_digest":"sha256:beab5423f48540fa1d5b40f8fefd86a51bf7b8e880f35a92fe1628b17d19775c","observation_id":"983a31b4-e4ab-4f69-af0b-3058fad203f5","resolution":{"observed_at":"2026-06-30T22:15:05.629765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.20564/citation-record","integrity":"/paper/2505.20564/integrity","json":"/paper/2505.20564/citation-record.json","paper":"/paper/2505.20564"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:37.313376Z","title":"While notable progress has been made in speech processing, African languages – including our focus languages, Igbo, Hausa, and Yoruba – have largely been left behind [ 7, 8]","venue":null,"work_id":"efdb0c62-8cab-4268-93bb-d974f3613b19","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:25.773223Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:af9948dd0ec838e09fa3320df1a21d41cada4d6a6e6220466bc6ece1042d0f99","observation_id":"11c1482e-737f-49a5-b229-e89af5940eed","resolution":{"observed_at":"2026-08-07T13:56:37.383518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.20564","snapshot_observed_at":"2026-08-07T13:56:25.830154Z","title":"Figure 1 illustrates our pipeline for creating the NaijaV oices dataset","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:25.830154Z"},"links":{"cited_paper":"/paper/2505.20564","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:3d551e92d4d9e7515ce8df320831fed4438fa7ab3b151dfe79a5e1bdf6a46bd6","observation_id":"3d6feb57-e0c6-4767-b7e1-11026b590d00","resolution":{"observed_at":"2026-08-07T13:56:25.830154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:36.968978Z","title":"It features a wide range of speech patterns influenced by age, education levels, accents, and speaking styles – from broken to formal speech, ethnic and dialectal influences","venue":null,"work_id":"abfbb007-ab90-4991-a178-5a491d5fe7be","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:25.900230Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:afc523fb6361e6d9b4ee8abb17f45853ffb9c85f65eba77db83812ab67c1850d","observation_id":"9af524c3-5de4-4e79-9692-67bc1d436ff8","resolution":{"observed_at":"2026-08-07T13:56:37.166709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0243.7847","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:30.767287Z","title":"Concretely, we finetune three selected ASR models on our dataset and evaluate them on both our test set (NV Test) and the FLEURS test set [20]","venue":null,"work_id":"8d10ad95-bd4a-4f5b-b669-6b148bb81c65","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:25.988475Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:6d7edffea7c1846bcdf1f68cb32517b8338b3b331d49cfeb896c5ded88469a4e","observation_id":"8ed4b76b-fcba-4576-a741-80f39a82d5fa","resolution":{"observed_at":"2026-08-07T13:56:30.806189Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:36.789630Z","title":"Built on the principles of ‘data farming’, our approach fosters a symbiotic relationship with language communities","venue":null,"work_id":"c3b44a18-dd77-439b-bd77-d5db343f13bd","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.055703Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:9de84e6466c66f03ec34e10d4174645915ddbb2f62cbd33c299ce1b081ceb931","observation_id":"203c6a02-ac4b-4ebf-bff6-8e7b517a217f","resolution":{"observed_at":"2026-08-07T13:56:36.866653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:36.613314Z","title":null,"venue":null,"work_id":"cb9a597d-b5c9-4a69-84ec-191aabb6920e","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.126261Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:9bb6c30da9cfb3a2d67393e48261929a5cb420531884022699259e92d5cf8440","observation_id":"5db42c02-a711-4017-bd61-39da9762e4ee","resolution":{"observed_at":"2026-08-07T13:56:36.719269Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:36.418938Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"82ceff08-890b-4cbc-b7cc-2edbe543e2ea","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.179296Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:c10773cd45f9aab14244f5f111550f3e5df0e7ba03550a1cf1c77fdf48fa70eb","observation_id":"bb12c951-1bbc-4dd9-9b29-8f898acf4929","resolution":{"observed_at":"2026-08-07T13:56:36.513465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:36.142263Z","title":"Scaling speech technology to 1,000+ languages,","venue":null,"work_id":"7b572387-11f7-4673-95a8-314415485288","year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.261837Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:f1abe2da163063520bbf12ce1ed658a3da6ebe4bbb096b38a2c8d4974e883e95","observation_id":"f421c7e7-1285-4a25-b393-dc14c5d0898b","resolution":{"observed_at":"2026-08-07T13:56:36.276579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16787","last_updated":"2023-11-04T19:10:06Z","snapshot_observed_at":"2026-07-06T16:38:29.188225Z","submitted_at":"2023-10-25T17:20:26Z","title":"The Data Provenance Initiative: A Large Scale Audit of Dataset Licensing & Attribution in AI","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16787","snapshot_observed_at":"2026-08-07T13:56:26.333107Z","title":"The data provenance initiative: A large scale audit of dataset licensing & attribution in AI,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.333107Z"},"links":{"cited_paper":"/paper/2310.16787","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:56e66a3616b2d607118a3f6118a7d9d5939da075116d7c9fae923b8a032068ee","observation_id":"83d1bb78-5b5f-407a-81cd-9c9a17bc186b","resolution":{"observed_at":"2026-08-07T13:56:26.333107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:35.941212Z","title":"IndicVoices: Towards building an inclusive multilingual speech dataset for Indian languages,","venue":null,"work_id":"cf836ac7-c1fb-4c16-aa35-8ef865133120","year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.416637Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:d0441b74049a1f7a7e17beddccd4795be9945b4ad16bcd0e4323c5161aeb3712","observation_id":"44c787d9-56e3-4010-9eee-db697332afbe","resolution":{"observed_at":"2026-08-07T13:56:36.034575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:35.739535Z","title":"IndicV oices-R: Unlocking a massive multilingual multi- speaker speech corpus for scaling indian TTS,","venue":null,"work_id":"c107cfd0-c9db-446b-84cd-d17929c8df67","year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.464410Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:ddc0de7ded701dfd905019bac211a743e4fc58d50545db97d5d018a0078c18e5","observation_id":"e1b1d2cb-26c7-4fae-b3ce-93280571f454","resolution":{"observed_at":"2026-08-07T13:56:35.846060Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08093","last_updated":"2024-02-15T18:57:26Z","snapshot_observed_at":"2026-07-06T17:29:13.310354Z","submitted_at":"2024-02-12T22:21:30Z","title":"BASE TTS: Lessons from building a billion-parameter Text-to-Speech model on 100K hours of data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08093","snapshot_observed_at":"2026-08-07T13:56:26.515597Z","title":"BASE TTS: Lessons from building a billion-parameter text-to- speech model on 100k hours of data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.515597Z"},"links":{"cited_paper":"/paper/2402.08093","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:bb6956a968c90b096cd20c2f35e141ad58c70bbcd063c1d79c43743e9dbf23a3","observation_id":"89b175f2-8933-40e3-a822-0401697b2d10","resolution":{"observed_at":"2026-08-07T13:56:26.515597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.7910/dvn/rxbncz","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:30.280208Z","title":"Replication data for Igbo Natural Language Processing Tasks I,","venue":null,"work_id":"7277291b-94c1-4532-bf6a-59e088662cf4","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.604471Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:c5570fb81de1b8e4c63abaf6337a1c163fbae22f5621ac064cd5e764f1bab5c2","observation_id":"b8e8a9e4-8227-4f6c-a7af-ae01145eebfa","resolution":{"observed_at":"2026-08-07T13:56:30.348444Z","resolver_source":"doi_truncated","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:35.526754Z","title":"Multi- lingual self-supervised speech representations improve the speech recognition of low-resource African languages with codeswitching,","venue":null,"work_id":"f7c4eb63-0849-4073-9704-698de3e3a2cb","year":2023},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.691212Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:f9c8d804af08709882f9577dea2f5ca753ec985e3784200be12d2a700b6d1a32","observation_id":"bdfb1d8e-de7e-46a0-b35f-567ccf5dbe1a","resolution":{"observed_at":"2026-08-07T13:56:35.621338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2003.11529","last_updated":"2020-03-13T09:01:02Z","snapshot_observed_at":"2026-08-03T19:02:55.352340Z","submitted_at":"2020-03-13T09:01:02Z","title":"Masakhane -- Machine Translation For Africa","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2003.11529","snapshot_observed_at":"2026-08-07T13:56:26.796632Z","title":"Masakhane - machine translation for Africa,","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.796632Z"},"links":{"cited_paper":"/paper/2003.11529","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:b5dfc73617b39021fb989260f5cfb90c9e933cfc4659976e268a8ea2004a9994","observation_id":"9830b3b2-4efb-4e9c-95c7-851e53957cb6","resolution":{"observed_at":"2026-08-07T13:56:26.796632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:35.300151Z","title":"Partici- patory research for low-resourced machine translation: A case study in African languages,","venue":null,"work_id":"9ceb26cf-79b3-4741-aea3-d4dfe8c8ff5b","year":2020},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:26.930277Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:768596d28659fd04c70d07b1f49a232d8892466b2b52008eed5595105a784cdb","observation_id":"a4114642-5a1d-424e-89c3-e0f5d75afd6b","resolution":{"observed_at":"2026-08-07T13:56:35.407955Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:35.104164Z","title":"The state and fate of linguistic diversity and inclusion in the NLP world,","venue":null,"work_id":"c26ef517-67f9-4c05-b423-93742dec73e7","year":2020},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.058131Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:a26ea6577e2aed0d09bdd9fa1ae152e3bcb142d5018e8f4c890a25da9b534461","observation_id":"4ad9046d-5454-4b94-961f-eb494a51f6a6","resolution":{"observed_at":"2026-08-07T13:56:35.176512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:34.901752Z","title":"A few thousand trans- lations go a long way! Leveraging pre-trained models for African news translation,","venue":null,"work_id":"a4cebe50-67d3-4d70-a9a4-e2fd126d1d9c","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.228195Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:d1f50abc54d74ac16131448408a488a089e5d4a0fe4692e6054b47ef1dc70650","observation_id":"e8546569-57eb-4caa-8ebf-9c30d18b0c80","resolution":{"observed_at":"2026-08-07T13:56:35.026735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:34.654567Z","title":null,"venue":null,"work_id":"f649a297-3486-4793-9588-ec4f4fce1773","year":2020},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.343643Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:318156db260861e47be626c35b83ab434adc382861d7b33bc9034dcbc4639359","observation_id":"cffa84ed-3bc5-4ca7-bc9d-1a9b04888149","resolution":{"observed_at":"2026-08-07T13:56:34.803033Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:34.447549Z","title":"GlobalPhone: A multi- lingual text & speech database in 20 languages,","venue":null,"work_id":"bbfd1735-8ed3-4188-a17f-8caf95e04bb2","year":2013},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.466576Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:15d009332d30b31001f0515ecfb2ed1603879fdc991485ad5e0e88c17af3442c","observation_id":"f9a5659e-f323-45da-ad3a-76927b2cf504","resolution":{"observed_at":"2026-08-07T13:56:34.531666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:34.231896Z","title":"YFACC: A Yor`ub´a speech–image dataset for cross-lingual keyword localisation through visual grounding,","venue":null,"work_id":"9d310cda-b145-423a-84be-392ce097803e","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.610794Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:7ae8b1e9e7df74f58721b004222313f02c893cd9c547da637daf44c3bf644d69","observation_id":"99dd0dba-51ca-4205-97d5-99d22f6feba7","resolution":{"observed_at":"2026-08-07T13:56:34.335515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16071","last_updated":"2024-03-27T08:56:01Z","snapshot_observed_at":"2026-07-06T16:00:12.721132Z","submitted_at":"2023-07-29T20:42:50Z","title":"\\`{I}r\\`{o}y\\`{i}nSpeech: A multi-purpose Yor\\`{u}b\\'{a} Speech Corpus","version":2},"cited_work":{"arxiv_id":"2307.16071","doi":null,"metadata_source":"pith","pith_arxiv_id":"2307.16071","snapshot_observed_at":"2026-08-07T13:56:30.555256Z","title":"\\`{I}r\\`{o}y\\`{i}nSpeech: A multi-purpose Yor\\`{u}b\\'{a} Speech Corpus","venue":"cs.CL","work_id":"f02a84a8-c00c-4b3e-b255-34ecb674230d","year":2023},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.728384Z"},"links":{"cited_paper":"/paper/2307.16071","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:6ec4cf865ea12bb76c3d2885c908e41ebfb28b6e922949f32e60593b6d36ec44","observation_id":"9c46c67d-597e-4272-8e55-8bb5173a78a1","resolution":{"observed_at":"2026-08-07T13:56:30.593979Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:34.033323Z","title":"V oices Unheard: NLP resources and models for Yor`ub´a regional dialects,","venue":null,"work_id":"35576404-d2c3-4eef-9f0f-0d41de033424","year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.834694Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:23b5fe56934879e89fbdfaf2572f9b9131fc3f85a81ec979259a0e2d1ed8c182","observation_id":"305af219-7afc-4346-a0dd-8610269909f9","resolution":{"observed_at":"2026-08-07T13:56:34.128670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:33.808573Z","title":"BibleTTS: a large, high-fidelity, multilingual, and uniquely African speech corpus,","venue":null,"work_id":"7275304b-478d-4c44-ae2a-99b170754af8","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:27.942667Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:20ebbf2d592047fa590cab94a63a71339acd6a20fd01675051939eff68d43a6c","observation_id":"0a70f040-138c-4672-aea9-90a2801964fa","resolution":{"observed_at":"2026-08-07T13:56:33.929958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:33.606421Z","title":"Common V oice: A massively-multilingual speech corpus,","venue":null,"work_id":"2bb1ab9d-d120-4312-87b9-3fb773c5e0e5","year":2020},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.035532Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:c4e93eee2708faf7b467d624273cc74b7ccee9f7e050c34fc0e12aeb086eaf6c","observation_id":"3147b531-225a-4bb9-ace9-2a0d06c27f34","resolution":{"observed_at":"2026-08-07T13:56:33.713462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:33.421349Z","title":"FLEURS: Few-shot learning evaluation of universal representations of speech,","venue":null,"work_id":"f7a67b91-319c-4247-9000-a9a6c68d77bd","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.199221Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:2ec71d879104ad98563ab390efcb3f2b24e2b7dbb76d880a32a192ccd781fde0","observation_id":"a7193215-c851-4cc5-ba8c-f30b9d110db3","resolution":{"observed_at":"2026-08-07T13:56:33.504591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:33.260744Z","title":"Quality at a glance: An audit of web-crawled multilingual datasets,","venue":null,"work_id":"45bab66f-4516-4ade-b236-ebc2d1bea414","year":2021},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.332611Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:f268448073d658da04bb7bcdfc4ca108e417c22c7ba8ad9e8c46654a2eaa5127","observation_id":"cc6f6625-dab4-47ea-9537-5f679ec36301","resolution":{"observed_at":"2026-08-07T13:56:33.338427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:33.064834Z","title":"Separating grains from the chaff: Using data filtering to improve multilingual translation for low-resourced African languages,","venue":null,"work_id":"2b5cb748-7c3d-4116-9591-671cf3169b4f","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.490123Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:b05e9e5ffedb20e9330aa1801fb91541d6a63f5ae7f84ee16e988314062dbd61","observation_id":"96873929-315a-446b-b9c0-09bbd1e25e6c","resolution":{"observed_at":"2026-08-07T13:56:33.149536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17557","last_updated":"2024-10-31T11:37:49Z","snapshot_observed_at":"2026-08-02T15:25:02.551919Z","submitted_at":"2024-06-25T13:50:56Z","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17557","snapshot_observed_at":"2026-08-07T13:56:28.580038Z","title":"The FineWeb Datasets: Decanting the web for the finest text data at scale,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.580038Z"},"links":{"cited_paper":"/paper/2406.17557","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:17251c97dca331eb8402a55d15829a1b385c56f0064368c563b957a024d9303a","observation_id":"8bc08eac-f04b-4a3f-8b00-637b31757013","resolution":{"observed_at":"2026-08-07T13:56:28.580038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:32.900999Z","title":"JW300: A wide-coverage parallel corpus for low-resource languages,","venue":null,"work_id":"e066b79f-9e88-4f19-b89f-b851cb8cd3f3","year":2019},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.668877Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:5cef7c651ff4266d6e82e1188dc99a67c120ee636bb73fa4f41f4f6bf600838c","observation_id":"acc0dc36-868a-438c-abd2-87d3171859db","resolution":{"observed_at":"2026-08-07T13:56:32.979742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:32.722691Z","title":"`Ir`oy`ınspeech: Yor`ub´a speech corpus,","venue":null,"work_id":"c4258008-ac73-4c72-b31a-ebd572bff757","year":2023},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.747396Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:6c1f06114b07b8f4fac41c55430e520f96854f7f76f3e50da1c8b986880407b8","observation_id":"b7d45a26-68e8-4323-a7b1-ff71d126d160","resolution":{"observed_at":"2026-08-07T13:56:32.809190Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.17632/j6kjmfrbby.2","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:30.114360Z","title":"Hausa speech corpus,","venue":null,"work_id":"56ee9f68-4c65-4d95-82a2-f5130465e808","year":2021},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.816780Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:11350e79376cea4f9dd0eca46053971036aed37d74bf73cc749051f3c1924595","observation_id":"26073f54-6068-493c-ac4a-fe7e5f0122ef","resolution":{"observed_at":"2026-08-07T13:56:30.221593Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:32.325779Z","title":"Kencorpus: A Kenyan language corpus of Swahili, Dholuo and Luhya for natural language processing tasks,","venue":null,"work_id":"e9d294fa-0fd7-40f0-92aa-be887b846c0d","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:28.892248Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:81fd51e363200820b6dbfd5ae46ec2b5eb4b32035df5a42d044e762b475c551a","observation_id":"c7af0928-5d8e-4744-88fe-e08c336cb797","resolution":{"observed_at":"2026-08-07T13:56:32.526703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:32.082472Z","title":"LIII. On lines and planes of closest fit to systems of points in space,","venue":null,"work_id":"1f6dec76-cf8a-42aa-bce2-7b26beaf28bf","year":1901},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.019144Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:019050b3eb08f0dee861bbc76a1b96c8e7387ace9fceb71148f683f518c4043e","observation_id":"2598dc06-a744-406e-8ddd-f238606782fa","resolution":{"observed_at":"2026-08-07T13:56:32.254430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:31.757532Z","title":"Robust signal-to-noise ratio estima- tion based on waveform amplitude distribution analysis,","venue":null,"work_id":"c683fca3-165f-45ad-80cf-d90aedd92166","year":2008},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.122151Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:57b82f112d43f612ff2fd134775acd2e2df296c32938d432e161892571f0f990","observation_id":"cd5e4632-baf8-4fb5-b79a-656297daf12e","resolution":{"observed_at":"2026-08-07T13:56:31.952490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:31.541408Z","title":"Audio quality feature,","venue":null,"work_id":"21b97921-7827-44f8-aea9-8fb0ec6dc499","year":2025},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.220939Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:52f859446252c18ea8d680e08570cf629ac05c4f81cc815683ab2e20cccbca73","observation_id":"5cca4861-c392-4c9a-b371-1c33e94ae017","resolution":{"observed_at":"2026-08-07T13:56:31.618588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:31.247842Z","title":"(n.d.) Evaluation","venue":null,"work_id":"e2ef1861-630e-4bc4-b2b1-f51b19f93d76","year":null},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.281822Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:d1d2c586e7ca39981535037b687a33282a62a786b114324c061e64f2e601c2d6","observation_id":"a364f504-22f4-479d-bc69-db6c22758a11","resolution":{"observed_at":"2026-08-07T13:56:31.388331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:31.094343Z","title":"Unsupervised cross-lingual representation learning for speech recognition,","venue":null,"work_id":"3af3996f-6e0a-418f-a7c2-0968709c9f93","year":2021},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.372393Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:d37625c4e8eed70cd0da60ec23fe52298d293b7e631572e5006f75af18f1ad54","observation_id":"8f7bbb81-62db-4296-bed6-c80cd69f1d15","resolution":{"observed_at":"2026-08-07T13:56:31.170421Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05187","last_updated":"2023-12-08T17:18:42Z","snapshot_observed_at":"2026-08-06T14:36:57.042438Z","submitted_at":"2023-12-08T17:18:42Z","title":"Seamless: Multilingual Expressive and Streaming Speech Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05187","snapshot_observed_at":"2026-08-07T13:56:29.523320Z","title":"Seamless: Multilingual expressive and streaming speech translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.523320Z"},"links":{"cited_paper":"/paper/2312.05187","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:91aad9cf686d8c482ffd70dfa36e890d264ca7bf949e0c4a066c1f081160c892","observation_id":"6dbcbbbd-4370-4a97-9aba-c82c3a6a2db4","resolution":{"observed_at":"2026-08-07T13:56:29.523320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:30.985154Z","title":"Small Data? No Problem! Ex- ploring the viability of pretrained multilingual language mod- els for low-resourced languages,","venue":null,"work_id":"d71a7bc3-0140-4b8d-9ebc-f8a4cca26305","year":2021},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.620364Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:f3c989b1c1372841a666af21917a7f7024e8da0a7683458eb74cf6f812b39892","observation_id":"b289d64b-ca04-497b-936c-3f7973f29acd","resolution":{"observed_at":"2026-08-07T13:56:31.052468Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.06409","last_updated":"2022-12-26T13:27:51Z","snapshot_observed_at":"2026-08-09T16:09:48.717980Z","submitted_at":"2021-12-13T03:57:36Z","title":"Data Collection and Quality Challenges in Deep Learning: A Data-Centric AI Perspective","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.06409","snapshot_observed_at":"2026-08-07T13:56:29.695164Z","title":"Data collection and quality challenges in deep learning: A data-centric ai perspective,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.695164Z"},"links":{"cited_paper":"/paper/2112.06409","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:c0c71373a9222bc969beaf3e7dbdce9ebf4d8494ecea8cfad21a0a60a1b8e082","observation_id":"afc908dd-cfef-403e-8a61-5063e67c5c35","resolution":{"observed_at":"2026-08-07T13:56:29.695164Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T13:56:30.869133Z","title":"What makes a high-quality training dataset for large language mod- els: A practitioners’ perspective,","venue":null,"work_id":"c0d5d2d4-c7b6-4872-b6e7-43c02cfc1e42","year":2024},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.985902Z"},"links":{"citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:87da8c08b88e3daeb5b3f46e3edf44e53e3290108c1b672914cc10db91c430cc","observation_id":"1adb9986-0c8f-4d73-b305-abffaafe2324","resolution":{"observed_at":"2026-08-07T13:56:30.929427Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.06404","last_updated":"2022-03-12T10:50:13Z","snapshot_observed_at":"2026-08-09T05:55:07.681135Z","submitted_at":"2022-03-12T10:50:13Z","title":"A Proposal to Study \"Is High Quality Data All We Need?\"","version":1},"cited_work":{"arxiv_id":"2203.06404","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.06404","snapshot_observed_at":"2026-08-07T13:56:30.415786Z","title":"A Proposal to Study \"Is High Quality Data All We Need?\"","venue":"cs.LG","work_id":"f21ad404-52e1-4d5f-b7ab-e4c2a873ee29","year":2022},"citing_paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages","version":3},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-07T13:56:29.888274Z"},"links":{"cited_paper":"/paper/2203.06404","citing_paper":"/paper/2505.20564"},"observation_digest":"sha256:3f1838136c4d1412904dd640c5d323ec9ab9a72a7570e94af5a0aaf065111509","observation_id":"88dd9e44-4744-41c2-b976-6488a62e9e9c","resolution":{"observed_at":"2026-08-07T13:56:30.468477Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.20564","last_updated":"2025-07-12T04:42:21Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T10:53:23.267759Z","submitted_at":"2025-05-26T22:53:48Z","title":"The NaijaVoices Dataset: Cultivating Large-Scale, High-Quality, Culturally-Rich Speech Data for African Languages"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":9,"verified_exact":3,"verified_fuzzy":29},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 3 inbound Pith citation observations for arXiv:2505.20564."}