{"as_of":"2026-08-24T02:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3c5e3cac150542e1fe69f035c2f862cf930391ee24c5871f681c3b34af642fd7","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":25,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:38:53.970993Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:10:07.966776Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-12T20:13:57.274992Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13577","last_updated":"2024-11-26T09:20:48Z","snapshot_observed_at":"2026-08-20T13:09:54.981053Z","submitted_at":"2024-11-15T04:16:45Z","title":"WavChat: A Survey of Spoken Dialogue Models","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-12T20:13:57.274992Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2411.13577"},"observation_digest":"sha256:6105b17224be110b54c2087fb620e09cfcf5c57dd6d71edf1a8fb168761f0556","observation_id":"424a669b-2016-4d62-a083-84d59e362c72","resolution":{"observed_at":"2026-08-12T20:13:57.274992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-16T12:38:53.970993Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2504.12254","last_updated":"2025-04-19T09:55:40Z","snapshot_observed_at":"2026-08-19T06:03:07.243471Z","submitted_at":"2025-04-16T17:05:14Z","title":"Advancing Arabic Speech Recognition Through Large-Scale Weakly Supervised Learning","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T12:38:53.970993Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2504.12254"},"observation_digest":"sha256:d9d5849ab7f8ed8cfb6dabb8e2a4f1918dfa43da5d121e3307a19897bc108ce2","observation_id":"a0686dc0-edd1-46c0-99f4-d5bd525bcb52","resolution":{"observed_at":"2026-08-16T12:38:53.970993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-07T14:12:48.463964Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19774","last_updated":"2025-05-26T09:57:59Z","snapshot_observed_at":"2026-08-17T12:35:12.313406Z","submitted_at":"2025-05-26T09:57:59Z","title":"DuRep: Dual-Mode Speech Representation Learning via ASR-Aware Distillation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:12:48.463964Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2505.19774"},"observation_digest":"sha256:9024179700cd63c9120dd67c10ee089a36adfa20c727a9fee382a53053323ec3","observation_id":"52b343b8-f89f-43d4-9dfa-e35a1fa2c00c","resolution":{"observed_at":"2026-08-07T14:12:48.463964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-07T13:49:20.652179Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.21578","last_updated":"2025-05-27T08:40:28Z","snapshot_observed_at":"2026-08-15T05:38:42.941028Z","submitted_at":"2025-05-27T08:40:28Z","title":"Loquacious Set: 25,000 Hours of Transcribed and Diverse English Speech Recognition Data for Research and Commercial Use","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:49:20.652179Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2505.21578"},"observation_digest":"sha256:48ae38e9429a9f696918499557fc179a1fae4ab0f5c42d0dff52c7716ee6be29","observation_id":"d94107cd-9278-4f9c-abfd-72870ccb72d7","resolution":{"observed_at":"2026-08-07T13:49:20.652179Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-07T12:12:19.512688Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.00338","last_updated":"2025-05-31T01:44:44Z","snapshot_observed_at":"2026-08-22T02:29:37.902053Z","submitted_at":"2025-05-31T01:44:44Z","title":"OWSM v4: Improving Open Whisper-Style Speech Models via Data Scaling and Cleaning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:12:19.512688Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2506.00338"},"observation_digest":"sha256:9e60f8e7a9af247c11f84557d25292675df2238f1f558c32bbbc375bdee484f7","observation_id":"9ec2465d-b686-4696-a457-74c0a22ec0aa","resolution":{"observed_at":"2026-08-07T12:12:19.512688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-07T11:52:13.381090Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for com- mercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01439","last_updated":"2025-06-02T08:52:50Z","snapshot_observed_at":"2026-08-21T16:01:55.909424Z","submitted_at":"2025-06-02T08:52:50Z","title":"Whale: Large-Scale multilingual ASR model with w2v-BERT and E-Branchformer with large speech data","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:13.381090Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2506.01439"},"observation_digest":"sha256:ec4aab9c315ef45236df232bb570039f3435ef70a9fd1bd10b9958242f47323d","observation_id":"7b443f9f-0199-4440-9394-983dbf2e8ece","resolution":{"observed_at":"2026-08-07T11:52:13.381090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-07T04:58:07.524656Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage.arXiv preprint arXiv:2111.09344,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09344","last_updated":"2025-06-11T02:50:49Z","snapshot_observed_at":"2026-08-08T17:53:50.843512Z","submitted_at":"2025-06-11T02:50:49Z","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T04:58:07.524656Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2506.09344"},"observation_digest":"sha256:a3aa5f5c733832d332eb77f3c1fdc745e94cf037569b5a944ad4d0add2126527","observation_id":"007423dc-82e9-40c5-be88-0607471da412","resolution":{"observed_at":"2026-08-07T04:58:07.524656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-06T23:34:07.432916Z","title":"Preprint, arXiv:2111.09344","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.17019","last_updated":"2025-06-20T14:17:42Z","snapshot_observed_at":"2026-08-14T08:13:11.379123Z","submitted_at":"2025-06-20T14:17:42Z","title":"Instituto de Telecomunica\\c{c}\\~oes at IWSLT 2025: Aligning Small-Scale Speech and Language Models for Speech-to-Text Learning","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T23:34:07.432916Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2506.17019"},"observation_digest":"sha256:f55836ef9e346f2d3fbbd87f49bd4584d56839ef98c5a0da28ef525a999c5e5a","observation_id":"82dc9500-d820-4672-8a3b-a1b4f3ff12c7","resolution":{"observed_at":"2026-08-06T23:34:07.432916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-06T22:18:12.458369Z","title":"Galvez, G","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21990","last_updated":"2025-06-27T07:57:13Z","snapshot_observed_at":"2026-08-13T13:18:08.227597Z","submitted_at":"2025-06-27T07:57:13Z","title":"Analyzing and Fine-Tuning Whisper Models for Multilingual Pilot Speech Transcription in the Cockpit","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:18:12.458369Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2506.21990"},"observation_digest":"sha256:dffcd20eeabfe3a8fd03b258b3a8da61c65f1dfd52c9fd5982e0fb4a034a2aff","observation_id":"1f9a1363-5163-4ac3-aa36-633587a2a7a7","resolution":{"observed_at":"2026-08-06T22:18:12.458369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-06T15:02:02.830714Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21138","last_updated":"2025-07-22T23:57:11Z","snapshot_observed_at":"2026-08-09T08:43:35.449807Z","submitted_at":"2025-07-22T23:57:11Z","title":"TTS-1 Technical Report","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T15:02:02.830714Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2507.21138"},"observation_digest":"sha256:6d300b86db932981b9ab2cd1ce378e95314d34d4d3c16c5262dd30c2911fb067","observation_id":"1116d9ff-1bb6-45fe-8c3b-2d690506a98d","resolution":{"observed_at":"2026-08-06T15:02:02.830714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-05T12:07:21.415152Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.01939","last_updated":"2025-09-02T04:20:12Z","snapshot_observed_at":"2026-08-15T15:45:56.408224Z","submitted_at":"2025-09-02T04:20:12Z","title":"Group Relative Policy Optimization for Speech Recognition","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T12:07:21.415152Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2509.01939"},"observation_digest":"sha256:adc5a4110ab10e9bff0b56a1473840b8ac1317c9aae81a4e30860e787b048036","observation_id":"3401423f-1608-4c4e-bf02-fad5fe8224fa","resolution":{"observed_at":"2026-08-05T12:07:21.415152Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-05T10:53:38.926567Z","title":"CoRR abs/2111.09344 (2021), https://arxiv","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.05359","last_updated":"2025-09-03T18:11:53Z","snapshot_observed_at":"2026-08-12T19:15:34.938531Z","submitted_at":"2025-09-03T18:11:53Z","title":"An Empirical Analysis of Discrete Unit Representations in Speech Language Modeling Pre-training","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T10:53:38.926567Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2509.05359"},"observation_digest":"sha256:99f4757c942c18d0b3e8e755a4d74293c28dc572545d0047acc6bdf931defd6a","observation_id":"11f03851-e677-4215-b44e-f17e27cdb100","resolution":{"observed_at":"2026-08-05T10:53:38.926567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2509.22220","last_updated":"2026-04-13T11:56:11Z","snapshot_observed_at":"2026-08-11T16:37:46.886811Z","submitted_at":"2025-09-26T11:32:51Z","title":"StableToken: A Noise-Robust Semantic Speech Tokenizer for Resilient SpeechLLMs","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-18T12:57:04.450462Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2509.22220"},"observation_digest":"sha256:1f39edf0c0ac5b52eac984356cfa307cf0df2323412fc39661551c94c8906f2f","observation_id":"6f3f1d5d-1fc1-4078-aac8-50ad37a2817d","resolution":{"observed_at":"2026-05-18T13:01:24.397871Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-08-03T19:05:08.543038Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2512.02201","last_updated":"2026-06-09T07:25:27Z","snapshot_observed_at":"2026-08-15T21:19:29.438275Z","submitted_at":"2025-12-01T20:49:10Z","title":"Swivuriso: The South African Next Voices Multilingual Speech Dataset","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T19:05:08.543038Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2512.02201"},"observation_digest":"sha256:4d191395d7fe1a282847f47b126a492afc279ecbe2ee34c1a9698e135be67b1f","observation_id":"f08ae745-8c0c-45c9-8e5d-5c4ce0d8d8ab","resolution":{"observed_at":"2026-08-03T19:05:08.543038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2604.08003","last_updated":"2026-04-09T09:07:52Z","snapshot_observed_at":"2026-08-18T19:47:10.020230Z","submitted_at":"2026-04-09T09:07:52Z","title":"Rethinking Entropy Allocation in LLM-based ASR: Understanding the Dynamics between Speech Encoders and LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:06:50.408402Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2604.08003"},"observation_digest":"sha256:13b51da00dbebd109e0f50761818b0039690e481c6d11780f89786ddc0dab1ba","observation_id":"d7b854e1-cb19-4514-8f8b-5dff47f5ff84","resolution":{"observed_at":"2026-05-11T05:30:57.407031Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.06765","last_updated":"2026-05-07T17:59:56Z","snapshot_observed_at":"2026-08-11T13:23:12.680939Z","submitted_at":"2026-05-07T17:59:56Z","title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-11T01:03:09.942984Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.06765"},"observation_digest":"sha256:facc3faa77ee9932a673498506cda33d53948846c064cd48c940252eb7a399ff","observation_id":"bb045c2c-b638-4482-a623-086d0037e76f","resolution":{"observed_at":"2026-05-11T04:50:55.964075Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.09568","last_updated":"2026-05-25T00:50:21Z","snapshot_observed_at":"2026-08-15T07:50:56.634385Z","submitted_at":"2026-05-10T14:29:35Z","title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T02:34:01.865434Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.09568"},"observation_digest":"sha256:c7781cbeedc19f28d4db77ea5b32671ee6b4bea604830d2d582b86023415ecc2","observation_id":"068edde5-1384-476f-b6df-1f05db66a483","resolution":{"observed_at":"2026-05-12T07:31:27.811824Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.09568","last_updated":"2026-05-25T00:50:21Z","snapshot_observed_at":"2026-08-15T07:50:56.634385Z","submitted_at":"2026-05-10T14:29:35Z","title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T22:57:59.897674Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.09568"},"observation_digest":"sha256:39cac84f2868ec8e6602f7e3dcb6b6f6141c943e8c35d292c5ebab42dbd54d30","observation_id":"be8a2da7-1b27-480e-87ce-879cc3e59f4b","resolution":{"observed_at":"2026-05-20T22:59:11.692624Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.09568","last_updated":"2026-05-25T00:50:21Z","snapshot_observed_at":"2026-08-15T07:50:56.634385Z","submitted_at":"2026-05-10T14:29:35Z","title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-30T22:53:18.321056Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.09568"},"observation_digest":"sha256:5864a55f7311c4da3a245fca62a9483a75ac670dea9601d4c4ff11c31bc330c1","observation_id":"49cb625b-4304-4618-9094-15bb842ea5f8","resolution":{"observed_at":"2026-07-01T13:45:45.419554Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.12387","last_updated":"2026-05-12T16:50:54Z","snapshot_observed_at":"2026-08-11T13:09:27.378345Z","submitted_at":"2026-05-12T16:50:54Z","title":"A Semi-Supervised Framework for Speech Confidence Detection using Whisper","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T04:00:49.889795Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.12387"},"observation_digest":"sha256:83d3b3eba0231ec6d94c6bcc97902c5409d16dace522bd5a5cd74b03f160431f","observation_id":"167ab0d4-6cc9-48fb-9d09-f00e78cea41b","resolution":{"observed_at":"2026-05-13T04:02:13.237407Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2605.20830","last_updated":"2026-05-20T07:21:36Z","snapshot_observed_at":"2026-08-17T06:26:22.986921Z","submitted_at":"2026-05-20T07:21:36Z","title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-21T02:32:26.122526Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2605.20830"},"observation_digest":"sha256:7e2dc6c8aa66169bb9a97f27759307aefe97e50a022d4bb5069e8f7c8a88727d","observation_id":"0d7c2b33-d648-40b7-8bcc-e3d509e1b74f","resolution":{"observed_at":"2026-05-21T02:33:55.507165Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2606.06837","last_updated":"2026-08-17T19:50:03Z","snapshot_observed_at":"2026-08-21T23:10:35.636931Z","submitted_at":"2026-06-05T02:24:19Z","title":"SEAM: Shortcut-Aware Real-Time Detection of Scripted vs. Spontaneous Speech for Interview Guardrails","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T21:19:56.932689Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2606.06837"},"observation_digest":"sha256:f111872f7ea6b3ec3e963a59d5e36344333f62f5772c57448aa0e02245172b3e","observation_id":"a8d9ac97-39d3-446e-bb53-a2f48a6c3582","resolution":{"observed_at":"2026-07-02T19:47:19.541629Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2606.22473","last_updated":"2026-06-21T12:33:44Z","snapshot_observed_at":"2026-08-14T14:26:37.762631Z","submitted_at":"2026-06-21T12:33:44Z","title":"Interleaved Speech Language Models Latently Work In Text","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T10:41:19.777779Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2606.22473"},"observation_digest":"sha256:88f9b0aa862c90ba101468622b037529cd434a57389be18d7ff6bf361fbafdaf","observation_id":"df4de4f9-255a-4315-90ca-f4423eb35d0b","resolution":{"observed_at":"2026-07-04T08:59:42.903183Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":"2111.09344","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-04T20:10:07.966776Z","title":"Galvez, G","venue":null,"work_id":"68c21db4-1050-44ca-822c-40033d58c0f3","year":2021},"citing_paper":{"arxiv_id":"2606.25391","last_updated":"2026-06-24T04:42:57Z","snapshot_observed_at":"2026-08-17T07:35:57.198246Z","submitted_at":"2026-06-24T04:42:57Z","title":"From Sounds to Scenes: A Benchmark for Evaluating Context-Aware Auditory Scene Understanding in Large Audio Language Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-06-25T20:36:24.901454Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2606.25391"},"observation_digest":"sha256:c3704c56aad851123c68255eccaaf057c066d49ff9a815a9c7e3b4406041c82f","observation_id":"6a8b02e2-f9f1-4c90-8e35-fb7561cdf825","resolution":{"observed_at":"2026-07-04T20:10:07.969024Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09344","snapshot_observed_at":"2026-07-11T09:35:44.381743Z","title":"The people’s speech: A large-scale diverse english speech recognition dataset for commercial usage,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.05051","last_updated":"2026-07-06T13:25:33Z","snapshot_observed_at":"2026-08-20T04:27:25.458314Z","submitted_at":"2026-07-06T13:25:33Z","title":"Listen, Think, Transcribe: Continuous Latent Test-Time Scaling for ASR","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-11T09:35:44.381743Z"},"links":{"cited_paper":"/paper/2111.09344","citing_paper":"/paper/2607.05051"},"observation_digest":"sha256:c4074698cfeb66aa7d132f6164687d036e2653ced220cbe9d40530f5aa81438f","observation_id":"b47761c7-f254-4d3a-9afa-934c9feee09d","resolution":{"observed_at":"2026-07-11T09:35:44.381743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2111.09344/citation-record","integrity":"/paper/2111.09344/integrity","json":"/paper/2111.09344/citation-record.json","paper":"/paper/2111.09344"},"outbound":[],"paper":{"arxiv_id":"2111.09344","last_updated":"2021-11-17T19:14:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-21T15:41:07.525153Z","submitted_at":"2021-11-17T19:14:40Z","title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 24 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 25 inbound Pith citation observations for arXiv:2111.09344."}