{"as_of":"2026-08-10T04:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a858a4627288aede0c5132b9b5c95b67494207e4a5aa5a31bc664bd8ab85a75a","coverage":[{"denominator":57,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":57,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:55:58.732055Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.16972/citation-record","integrity":"/paper/2505.16972/integrity","json":"/paper/2505.16972/citation-record.json","paper":"/paper/2505.16972"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:55:53.174990Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.174990Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:7f47457579b64bfe6205da5b53e491e407a87b0d72e39555a92ff37184d8534f","observation_id":"749b4bef-0a31-4856-a0bc-f8e3aeb8a981","resolution":{"observed_at":"2026-08-07T14:55:53.174990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:55:53.259170Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.259170Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:43afa06c50a7aa3bda90ff64c3efb5d30b2c773728dab9a88ca3043cd72f534f","observation_id":"d5a9b5a6-989a-4d16-b554-ddef40519e90","resolution":{"observed_at":"2026-08-07T14:55:53.259170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.872150Z","title":null,"venue":null,"work_id":"cb90b504-1348-429e-8e68-78a4b148d8a4","year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.443183Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:3668839e0e4e33e87e9178d3ac4ff1c6407819d0b109057c7f72fad9c80aebbb","observation_id":"b244d73b-3bb0-4fdb-88d5-9139b5b7122f","resolution":{"observed_at":"2026-08-07T14:56:03.915518Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.736558Z","title":null,"venue":null,"work_id":"1af1356d-ad9f-41c9-a3cb-fbebf9ade309","year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.583069Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:e5a052d03ef1a771665037e352e3bb1eafaf5641b7964ffd121211bd133c6377","observation_id":"d7b3252d-7557-4296-8eef-e011c7e64fa4","resolution":{"observed_at":"2026-08-07T14:56:03.815728Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06670","last_updated":"2020-03-05T20:37:08Z","snapshot_observed_at":"2026-08-05T11:17:11.092255Z","submitted_at":"2019-12-13T19:22:44Z","title":"Common Voice: A Massively-Multilingual Speech Corpus","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.06670","snapshot_observed_at":"2026-08-07T14:55:53.700409Z","title":"Tyers, and Gregor Weber","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.700409Z"},"links":{"cited_paper":"/paper/1912.06670","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:83a090189d7d55161719380ce70808f0cb6ed7c39382946be411e1b7c4e583ec","observation_id":"51065def-f9b0-4641-bc23-c1fe90918aba","resolution":{"observed_at":"2026-08-07T14:55:53.700409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08466","last_updated":"2023-04-17T17:42:29Z","snapshot_observed_at":"2026-07-06T15:16:38.500909Z","submitted_at":"2023-04-17T17:42:29Z","title":"Synthetic Data from Diffusion Models Improves ImageNet Classification","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08466","snapshot_observed_at":"2026-08-07T14:55:53.847905Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.847905Z"},"links":{"cited_paper":"/paper/2304.08466","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:178455ac10d479509ce01abe4e99f8646e98a8369a3e46bcabcde2b17d6c827e","observation_id":"2c8ad1b7-942e-4d74-90aa-0c1b497eb2c8","resolution":{"observed_at":"2026-08-07T14:55:53.847905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.619804Z","title":null,"venue":null,"work_id":"de2e276b-5b01-497d-a77a-4a13f5ba51c6","year":2021},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:53.982528Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:84ff61bbc91a39e93db01595aee498134fd69d478d456415e49986aa536807ba","observation_id":"07068995-31a3-45ad-8482-05f6f88c067f","resolution":{"observed_at":"2026-08-07T14:56:03.670409Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.483276Z","title":null,"venue":null,"work_id":"72a4fa95-f2b8-47a3-b73d-8d21a006a73f","year":2021},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.154721Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:75ec26fb74226cae865a9006a671b7469fc42a92c56febb7965e50c1ebe9f180","observation_id":"a52c2ee2-3e01-4230-88d4-43cd45c4dc80","resolution":{"observed_at":"2026-08-07T14:56:03.532208Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.11477","last_updated":"2020-10-22T06:09:10Z","snapshot_observed_at":"2026-08-07T05:13:16.889179Z","submitted_at":"2020-06-20T02:35:02Z","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.11477","snapshot_observed_at":"2026-08-07T14:55:54.314575Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.314575Z"},"links":{"cited_paper":"/paper/2006.11477","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:a987c7b2ee5e498c84a04864619f73c7fd285ecd1c81eb2d9567fbedf4f7dc8b","observation_id":"463240d4-c012-4afd-8079-e7ebc5705b0d","resolution":{"observed_at":"2026-08-07T14:55:54.314575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-07-06T18:42:05.670928Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-07T14:55:54.427250Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.427250Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:80feb92648c36ce83609c7f62b984282d41dc3bd3c797fdb77663e0cafa77e4a","observation_id":"888acafe-876e-438d-99b9-60c6a4d3cabd","resolution":{"observed_at":"2026-08-07T14:55:54.427250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.299136Z","title":null,"venue":null,"work_id":"3210786e-e82d-4297-96a7-07f4c4beec23","year":2019},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.498525Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:a46181423b4ce17482ab934e58899e164c20115459d468678ec70c34f0d2c7e2","observation_id":"15b3e51d-1b6c-483c-9af9-a277b079a49c","resolution":{"observed_at":"2026-08-07T14:56:03.383130Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.206845Z","title":null,"venue":null,"work_id":"4840eee4-f1f9-4343-8258-3586054de585","year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.566641Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:3d9c1c14b0c46d99cdb366b684a8ad1ed70eb8bfc42f26be12fa59abe5059211","observation_id":"35e732a5-8035-43a1-8e9c-66ff798c8af9","resolution":{"observed_at":"2026-08-07T14:56:03.244325Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04904","last_updated":"2024-06-07T12:56:11Z","snapshot_observed_at":"2026-08-06T13:29:11.244845Z","submitted_at":"2024-06-07T12:56:11Z","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04904","snapshot_observed_at":"2026-08-07T14:55:54.665380Z","title":"o lge, G \\","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.665380Z"},"links":{"cited_paper":"/paper/2406.04904","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:ced5b6c8c9e85d159ecf4fbd26441fbf8e5b7052cb02d79857e03ee636514341","observation_id":"fb412036-97d5-4785-9018-d1fe2474c66e","resolution":{"observed_at":"2026-08-07T14:55:54.665380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00837","last_updated":"2024-07-02T17:23:44Z","snapshot_observed_at":"2026-08-09T19:13:54.402948Z","submitted_at":"2024-06-30T21:40:26Z","title":"Towards Robust Speech Representation Learning for Thousands of Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00837","snapshot_observed_at":"2026-08-07T14:55:54.752647Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.752647Z"},"links":{"cited_paper":"/paper/2407.00837","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:070f8fb1d51a1f6209555a2fe827a7870a9f56b0611ea8c909bd2602718da637","observation_id":"9b65f5b2-3f11-49c6-800e-ef63fd2fe423","resolution":{"observed_at":"2026-08-07T14:55:54.752647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21646","last_updated":"2024-08-30T06:50:51Z","snapshot_observed_at":"2026-07-06T18:55:02.790902Z","submitted_at":"2024-07-31T14:48:27Z","title":"Towards Achieving Human Parity on End-to-end Simultaneous Speech Translation via LLM Agent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21646","snapshot_observed_at":"2026-08-07T14:55:54.842142Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.842142Z"},"links":{"cited_paper":"/paper/2407.21646","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:c0a335bc0ae1cbbc1201f33b98ac96f39a19c99e9a56eb6852a2504ded54b32e","observation_id":"ae97234f-0160-422c-af4e-9c536861be1f","resolution":{"observed_at":"2026-08-07T14:55:54.842142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05187","last_updated":"2023-12-08T17:18:42Z","snapshot_observed_at":"2026-08-06T14:36:57.042438Z","submitted_at":"2023-12-08T17:18:42Z","title":"Seamless: Multilingual Expressive and Streaming Speech Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05187","snapshot_observed_at":"2026-08-07T14:55:54.912281Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.912281Z"},"links":{"cited_paper":"/paper/2312.05187","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:8be98b22c80e615673e1967dfe4a24764a031894617e6b344136da50418ded2f","observation_id":"851d97c0-f3eb-4d2c-b8c6-bb67bb7072d4","resolution":{"observed_at":"2026-08-07T14:55:54.912281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:03.047833Z","title":null,"venue":null,"work_id":"fa06087d-95a7-40fd-80e3-da8ec96e1190","year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:54.987167Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:f6efcdd356e3fd4d269c9b0c0274af69e9a55364b3768e7b2fe927315c157738","observation_id":"f2b799f4-8ca8-453e-9fba-cd4eaae2f49d","resolution":{"observed_at":"2026-08-07T14:56:03.105039Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-07T14:55:55.065966Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.065966Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:5d5c0f02d2c870952da03151e74e86675a2a30573cc4ecf40f2637430a8f49f9","observation_id":"3c0638db-9725-4da7-8d96-67e4b858b084","resolution":{"observed_at":"2026-08-07T14:55:55.065966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.10097","last_updated":"2023-06-16T17:17:06Z","snapshot_observed_at":"2026-07-06T15:43:39.177848Z","submitted_at":"2023-06-16T17:17:06Z","title":"CML-TTS A Multilingual Dataset for Speech Synthesis in Low-Resource Languages","version":1},"cited_work":{"arxiv_id":"2306.10097","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.10097","snapshot_observed_at":"2026-08-07T14:55:59.705961Z","title":"CML-TTS A Multilingual Dataset for Speech Synthesis in Low-Resource Languages","venue":"eess.AS","work_id":"04585cb9-24c5-4193-a106-7b1b4df9c20c","year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.134893Z"},"links":{"cited_paper":"/paper/2306.10097","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:b3ac074b81a95171a9df541c2077d6c4665ac9ce5208b884548d1c4f196a779e","observation_id":"bb08feb4-246b-4e5e-b096-79b083b6457d","resolution":{"observed_at":"2026-08-07T14:55:59.809037Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.09381","last_updated":"2018-10-03T01:42:36Z","snapshot_observed_at":"2026-08-04T02:32:47.288584Z","submitted_at":"2018-08-28T16:05:40Z","title":"Understanding Back-Translation at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.09381","snapshot_observed_at":"2026-08-07T14:55:55.225525Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.225525Z"},"links":{"cited_paper":"/paper/1808.09381","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:4b88193054d855018a0289f593fe30b95e7a2833963c90086c67f53a2b124ffa","observation_id":"42604363-4e17-44da-9d12-57971fd4abbc","resolution":{"observed_at":"2026-08-07T14:55:55.225525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.917724Z","title":null,"venue":null,"work_id":"b18e7fe6-03bf-4bfd-82b8-06e1d2b9f998","year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.298928Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:984098f2a23220f6e965fe4e6f0c4d2aa74c93c487a3e5192871f28fa6af23e4","observation_id":"614ab78c-b34f-489d-b9a8-3d7585b2a781","resolution":{"observed_at":"2026-08-07T14:56:02.983278Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.763448Z","title":"Hasegawa-Johnson, Shiyu Chang, and Yang Zhang","venue":null,"work_id":"9d5b0443-f76a-4938-bd87-3e7f0ee13230","year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.388526Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:d65f9f9036aeb7d9a510669737b19338de62eef7ffc2eead9ea716ce32758b7f","observation_id":"d695a88a-0cf4-4a2e-9a08-59f3d2684eed","resolution":{"observed_at":"2026-08-07T14:56:02.806913Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11644","last_updated":"2023-10-02T06:12:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-20T16:14:25Z","title":"Textbooks Are All You Need","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11644","snapshot_observed_at":"2026-08-07T14:55:55.478048Z","title":"Shah, Harkirat Singh Behl, Xin Wang, S \\'e bastien Bubeck, Ronen Eldan, Adam Tauman Kalai, Yin Tat Lee, and Yuan-Fang Li","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.478048Z"},"links":{"cited_paper":"/paper/2306.11644","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:6e32c6a6bb81741922caf7adce9b1ee59fec528b95c4d85ee7c375eb05dabd74","observation_id":"d9558d3f-fb26-44cf-af03-08a9bafb6415","resolution":{"observed_at":"2026-08-07T14:55:55.478048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.644757Z","title":null,"venue":null,"work_id":"118aaf97-d4c7-4c92-8fc5-42daaf0d0773","year":2016},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.542956Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:465af933fff34f8a08da775b2bdf01fb6efc8f456d21df04a7c153cd0db9b09e","observation_id":"373baf00-f961-45f3-b761-91202f53915b","resolution":{"observed_at":"2026-08-07T14:56:02.684344Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05361","last_updated":"2024-09-07T15:08:24Z","snapshot_observed_at":"2026-08-07T13:19:22.496186Z","submitted_at":"2024-07-07T13:24:54Z","title":"Emilia: An Extensive, Multilingual, and Diverse Speech Dataset for Large-Scale Speech Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05361","snapshot_observed_at":"2026-08-07T14:55:55.604407Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.604407Z"},"links":{"cited_paper":"/paper/2407.05361","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:2e3b166d995202dc6a56006f547ba95367a5aa5703ef7cf407db112d2e502f46","observation_id":"aa6c3185-dcb8-4aa5-93ae-5dd8727a85ab","resolution":{"observed_at":"2026-08-07T14:55:55.604407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.425501Z","title":null,"venue":null,"work_id":"41f31885-bfc7-47cc-ae7b-9f8c192cd28d","year":2018},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.661496Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:ace430bac5846ba777dcd0cb35c8ddad2f1a21fca6ec7592f4084b9a416b6f8a","observation_id":"2a34c5b7-44a9-4d22-90ca-e3fe94a7f1f4","resolution":{"observed_at":"2026-08-07T14:56:02.482808Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13585","last_updated":"2023-12-21T05:32:49Z","snapshot_observed_at":"2026-08-09T02:04:31.523385Z","submitted_at":"2023-12-21T05:32:49Z","title":"Speech Translation with Large Language Models: An Industrial Practice","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13585","snapshot_observed_at":"2026-08-07T14:55:55.748608Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.748608Z"},"links":{"cited_paper":"/paper/2312.13585","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:8c5a398670bd6df168c1c58e1e78184dd0d7473c1ad874d31eb93873aa1ea331","observation_id":"edee7b2d-1094-48e3-9e5f-16779cfea925","resolution":{"observed_at":"2026-08-07T14:55:55.748608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.323982Z","title":null,"venue":null,"work_id":"23c0d6d3-a7f6-4524-a2ca-f90fbebaf380","year":2005},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.855819Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:60aa04e6d8442e52b0700bb6c48396b93ec66af3545eb9c2fe4644b580bd75ed","observation_id":"57c6908c-707b-4e18-a079-a69d3e6608b0","resolution":{"observed_at":"2026-08-07T14:56:02.381898Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.05646","last_updated":"2020-10-23T09:12:04Z","snapshot_observed_at":"2026-07-06T10:03:33.857220Z","submitted_at":"2020-10-12T12:33:43Z","title":"HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.05646","snapshot_observed_at":"2026-08-07T14:55:55.933535Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:55.933535Z"},"links":{"cited_paper":"/paper/2010.05646","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:69080dc2dc761e182891083b35cfa3dd4211c987c792c4af697bbe6f6930acb0","observation_id":"968f3035-3656-43ed-aa5d-cb8f91ee8ba5","resolution":{"observed_at":"2026-08-07T14:55:55.933535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.183969Z","title":null,"venue":null,"work_id":"a0e5d6bf-b8c3-4401-a9bc-443b93f31c2f","year":2018},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.043409Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:5c5cc8c71061d13d6e770e6618359bc5c1261247787053def9d0c9f019b06fda","observation_id":"b1a35c0c-15e6-4e3b-931c-20a8f00fe4df","resolution":{"observed_at":"2026-08-07T14:56:02.230660Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13064","last_updated":"2024-02-20T15:00:35Z","snapshot_observed_at":"2026-08-06T08:24:00.729497Z","submitted_at":"2024-02-20T15:00:35Z","title":"Synthetic Data (Almost) from Scratch: Generalized Instruction Tuning for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13064","snapshot_observed_at":"2026-08-07T14:55:56.193027Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.193027Z"},"links":{"cited_paper":"/paper/2402.13064","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:948e582122cdbb9d43aa59b5d030b5bc5c9410290bf84fad766fb49eacb764f4","observation_id":"704475c2-887a-4818-b96d-c4cba4812de1","resolution":{"observed_at":"2026-08-07T14:55:56.193027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:02.044293Z","title":null,"venue":null,"work_id":"ff29a24a-a6eb-44ad-8f1e-6573b8300496","year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.305477Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:5ec967d54604106c650bac609bfd3f914bc68ed1cd3ff9d769e3e7d79a2671e6","observation_id":"3d92d0db-9563-4a93-aa41-6c2dace948a2","resolution":{"observed_at":"2026-08-07T14:56:02.087331Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.03001","last_updated":"2020-07-08T03:02:06Z","snapshot_observed_at":"2026-08-06T20:03:11.649769Z","submitted_at":"2020-07-06T18:43:38Z","title":"Massively Multilingual ASR: 50 Languages, 1 Model, 1 Billion Parameters","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.03001","snapshot_observed_at":"2026-08-07T14:55:56.404903Z","title":"Hannun, Vitaliy Liptchinsky, Gabriel Synnaeve, and Ronan Collobert","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.404903Z"},"links":{"cited_paper":"/paper/2007.03001","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:a089d0cb8f97eca404592830aabc632490da7974b424abf0164f7c76bafa8c22","observation_id":"0a8853c1-bc77-4cc2-b259-aa41ef251803","resolution":{"observed_at":"2026-08-07T14:55:56.404903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13516","last_updated":"2023-05-22T22:09:41Z","snapshot_observed_at":"2026-08-02T15:37:26.541116Z","submitted_at":"2023-05-22T22:09:41Z","title":"Scaling Speech Technology to 1,000+ Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13516","snapshot_observed_at":"2026-08-07T14:55:56.498003Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.498003Z"},"links":{"cited_paper":"/paper/2305.13516","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:dcb0cecf00e816cfea1bad66aa63584ae55bc160745a276be3a62dd6e6bbd2b9","observation_id":"48978583-1258-41b0-8653-d9ddca2174a2","resolution":{"observed_at":"2026-08-07T14:55:56.498003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.03411","last_updated":"2020-12-19T09:18:21Z","snapshot_observed_at":"2026-07-06T10:21:14.277598Z","submitted_at":"2020-12-07T01:53:45Z","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.03411","snapshot_observed_at":"2026-08-07T14:55:56.615438Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.615438Z"},"links":{"cited_paper":"/paper/2012.03411","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:d5320b57bd56deb6a56d9b2166bd4450699612ad1a083d5a5d037e6cb9a7ca37","observation_id":"875e3c76-392e-4d58-b291-005c49d5375b","resolution":{"observed_at":"2026-08-07T14:55:56.615438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19674","last_updated":"2024-06-28T06:22:23Z","snapshot_observed_at":"2026-07-06T18:38:18.115901Z","submitted_at":"2024-06-28T06:22:23Z","title":"Less is More: Accurate Speech Recognition & Translation without Web-Scale Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19674","snapshot_observed_at":"2026-08-07T14:55:56.665186Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.665186Z"},"links":{"cited_paper":"/paper/2406.19674","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:f7c64efc7343466e1d835a774edd0ae50c38e6168ddd945c7d24fb42cabf3cc7","observation_id":"99b7891b-7077-4c39-b587-7e6c1e7f1550","resolution":{"observed_at":"2026-08-07T14:55:56.665186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-07T14:55:56.715907Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.715907Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:fd06c935ade6d9d1c2e3c854403d9e0ebaa7bd33614e86569752ec02ac8ed0f7","observation_id":"7489012e-c329-4f5a-8be7-8331d9ac307c","resolution":{"observed_at":"2026-08-07T14:55:56.715907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.883901Z","title":null,"venue":null,"work_id":"3821a0a3-22be-4869-a344-a0766e925b5b","year":2019},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.774284Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:5d1c2c74adc607b26bd25a4fb9cf9e12ed0939a5f6b626333f917daeb8104e34","observation_id":"eb7f802f-3fe0-4d57-babc-bef3dcafee9f","resolution":{"observed_at":"2026-08-07T14:56:01.957908Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.707024Z","title":"Blattmann, Dominik Lorenz, Patrick Esser, and Bj \\\"o rn Ommer","venue":null,"work_id":"a58416ed-050d-472e-8d99-c134a6900073","year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.858264Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:f9f7ce99e22fc4e9d57619ffd573536b7597fa65453b29c7a99c9d96f2d76901","observation_id":"259d6162-e9a0-4a07-bb3d-090db4b295ae","resolution":{"observed_at":"2026-08-07T14:56:01.810253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.513297Z","title":null,"venue":null,"work_id":"d06faeec-0d72-4989-bcc2-0ec5628f194f","year":2016},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:56.967930Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:bf066ded98ee726ffbd7b57aee1c3c5f024b76794b4b0fc8d7468f606b6f3430","observation_id":"31a069dc-96b6-4896-b176-9adfe6b9f8b1","resolution":{"observed_at":"2026-08-07T14:56:01.596909Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.331272Z","title":null,"venue":null,"work_id":"cec389ef-b23d-4e16-8d07-98d0f9a6744a","year":2016},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.112780Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:f37bb6b69c195835d4576422630672baf3b8b0ec8e6f32e8ae4a94fa3163060b","observation_id":"48d01bbb-d7de-4888-b977-9ccb38332a0e","resolution":{"observed_at":"2026-08-07T14:56:01.413813Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00814","last_updated":"2024-05-29T14:21:47Z","snapshot_observed_at":"2026-08-08T02:04:12.090464Z","submitted_at":"2023-06-01T15:40:32Z","title":"Vocos: Closing the gap between time-domain and Fourier-based neural vocoders for high-quality audio synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00814","snapshot_observed_at":"2026-08-07T14:55:57.198647Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.198647Z"},"links":{"cited_paper":"/paper/2306.00814","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:346ec7f484b5f995c941a6f5e74b68f55dab4638c3debd1179e48ea054fe04a0","observation_id":"fa95a800-6595-417b-af9a-f94e5bea44ef","resolution":{"observed_at":"2026-08-07T14:55:57.198647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.174007Z","title":null,"venue":null,"work_id":"feb75b3f-51a0-42e4-9170-544d93f2cee2","year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.312465Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:941f9521329731ff02efa5b07376c02866bfc3d54884fe60e1c22f75e4558c5a","observation_id":"3e8cffa8-06ad-475f-8c14-b3629bb6e103","resolution":{"observed_at":"2026-08-07T14:56:01.264342Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00984","last_updated":"2023-10-26T15:16:57Z","snapshot_observed_at":"2026-07-06T15:36:45.014921Z","submitted_at":"2023-06-01T17:59:51Z","title":"StableRep: Synthetic Images from Text-to-Image Models Make Strong Visual Representation Learners","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00984","snapshot_observed_at":"2026-08-07T14:55:57.433402Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.433402Z"},"links":{"cited_paper":"/paper/2306.00984","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:3e0289b7a18fa42e29bd95aa24ea92b7b32df114b68e51519f96628433b6050b","observation_id":"f7a78e29-80ab-4c3d-b9aa-f77f7ab8bf71","resolution":{"observed_at":"2026-08-07T14:55:57.433402Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-07T14:55:57.537044Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.537044Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:0e0b4819bcecae6ef92d846b4d2560b7e9479bcb83626cb5ec52e43b984291d0","observation_id":"6b8e309e-b3d5-48c3-8f9d-c4a615853a3d","resolution":{"observed_at":"2026-08-07T14:55:57.537044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T14:55:57.615225Z","title":"Stone, Peter Albert, Amjad Almahairi, Yasmine Babaei, Nikolay Bashlykov, Soumya Batra, Prajjwal Bhargava, Shruti Bhosale, Daniel M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.615225Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:42a75019088f0ec16ddb631fc68392e32b566b294f4ac700d4d99f2db9bcc54a","observation_id":"e9b37409-a26b-4b71-919c-2247ad31a3d0","resolution":{"observed_at":"2026-08-07T14:55:57.615225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.07944","last_updated":"2025-06-10T20:01:59Z","snapshot_observed_at":"2026-08-07T10:59:13.865982Z","submitted_at":"2023-02-07T20:42:28Z","title":"Effective Data Augmentation With Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.07944","snapshot_observed_at":"2026-08-07T14:55:57.729459Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.729459Z"},"links":{"cited_paper":"/paper/2302.07944","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:61ca5bf95a49d6096b5459174423abdfe876b89dd4306da9ae49aba741b26e46","observation_id":"1b39a7ae-1242-48e5-a95f-f55e7ee62b94","resolution":{"observed_at":"2026-08-07T14:55:57.729459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00390","last_updated":"2021-07-27T04:04:57Z","snapshot_observed_at":"2026-08-09T19:09:23.880235Z","submitted_at":"2021-01-02T07:24:21Z","title":"VoxPopuli: A Large-Scale Multilingual Speech Corpus for Representation Learning, Semi-Supervised Learning and Interpretation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00390","snapshot_observed_at":"2026-08-07T14:55:57.846781Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.846781Z"},"links":{"cited_paper":"/paper/2101.00390","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:e080ffb7f199120ad2dedc3812d6bdc8e7dc5b6ba567995b65918366148186d1","observation_id":"b12b32fd-de54-4556-817b-b2c42d11dca1","resolution":{"observed_at":"2026-08-07T14:55:57.846781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-07T14:55:57.998249Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:57.998249Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:bd6ff78d7b6f5bab9c3ca8f5255d2d95edc6e89d097deefe1ef9a215328465c4","observation_id":"9dd89f1e-ac5d-4aed-99c1-9262f5d8728a","resolution":{"observed_at":"2026-08-07T14:55:57.998249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:01.036845Z","title":null,"venue":null,"work_id":"a3cad4fc-ed1e-4906-9d7b-5ba9611e786b","year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.086002Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:521f82e45a786bf796e7c8ab4ad918da23af78db497ad0121cfea5349f3e18f8","observation_id":"14196f2b-ee01-4167-89f4-46fe289857e1","resolution":{"observed_at":"2026-08-07T14:56:01.063964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:00.832426Z","title":null,"venue":null,"work_id":"7a16ea5c-a339-4f7f-9e40-df0890d1a6d3","year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.222753Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:cc7440f150a893b1011efaba1e87dcdecbacdb716530d7d5f3917b7bbd1111f4","observation_id":"0b620978-fab8-4bfd-927c-dcf3d9c6e675","resolution":{"observed_at":"2026-08-07T14:56:00.913779Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:00.647692Z","title":null,"venue":null,"work_id":"ea46945d-8291-4063-a03c-017592585568","year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.293114Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:fc9e349d690b0374a7a8b81cd948c4dca018dd2937d9a1fda116a83f1013f197","observation_id":"cb572e1c-c837-40e5-b3bf-23355b8c2610","resolution":{"observed_at":"2026-08-07T14:56:00.730417Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19341","last_updated":"2023-10-30T08:31:47Z","snapshot_observed_at":"2026-08-06T06:09:43.812715Z","submitted_at":"2023-10-30T08:31:47Z","title":"Skywork: A More Open Bilingual Foundation Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19341","snapshot_observed_at":"2026-08-07T14:55:58.395604Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.395604Z"},"links":{"cited_paper":"/paper/2310.19341","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:ed41d1fab5b6971c16452007a12f4c3d938b84b4568b38128030d3dcf4cc9ea8","observation_id":"298d3a94-989d-42fc-ba0b-9a1891f70374","resolution":{"observed_at":"2026-08-07T14:55:58.395604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.16726","last_updated":"2024-10-22T06:25:16Z","snapshot_observed_at":"2026-08-06T00:05:25.787309Z","submitted_at":"2024-10-22T06:25:16Z","title":"Enhancing Low-Resource ASR through Versatile TTS: Bridging the Data Gap","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.16726","snapshot_observed_at":"2026-08-07T14:55:58.468368Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.468368Z"},"links":{"cited_paper":"/paper/2410.16726","citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:cae198218e52b35cacda960db2c9af39751179e0ed9cc80019528faafe07e8ed","observation_id":"6b6f0483-c670-46c6-8a84-3915bb604f78","resolution":{"observed_at":"2026-08-07T14:55:58.468368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:00.502306Z","title":null,"venue":null,"work_id":"2a5d9cc6-750b-40d3-92eb-9370bc315d91","year":2019},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.540112Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:8a38bf6357351f660514b82758a215629cf72c0564d1051cdf44402f5d5914f2","observation_id":"230d8c32-385a-4d69-bc79-a1f6c7215863","resolution":{"observed_at":"2026-08-07T14:56:00.581979Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:00.362796Z","title":null,"venue":null,"work_id":"ee333aad-61ad-41bd-941b-1d1cb043eba8","year":2022},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.638358Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:8a8fdc2861d0ca0b8b5a52888220ca12b41d6606f44865c5f17f224f31af84a1","observation_id":"ed7f5382-0d3b-4d65-a29d-a64967f2ff51","resolution":{"observed_at":"2026-08-07T14:56:00.411285Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:56:00.194412Z","title":null,"venue":null,"work_id":"5d7fa994-b1ae-4345-b59a-431f9ee54cec","year":2021},"citing_paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-07T14:55:58.732055Z"},"links":{"citing_paper":"/paper/2505.16972"},"observation_digest":"sha256:598fc21a37e54dd123ec11fb1813bbccb856eaaef7c9e9e78423dec93b0e83dc","observation_id":"04e01395-1449-4357-81d7-0b687c7b1a02","resolution":{"observed_at":"2026-08-07T14:56:00.263139Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.16972","last_updated":"2025-05-22T17:51:05Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T09:32:42.256193Z","submitted_at":"2025-05-22T17:51:05Z","title":"From Tens of Hours to Tens of Thousands: Scaling Back-Translation for Speech Recognition"},"reference_resolution":{"displayed":57,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":54,"verified_exact":1,"verified_fuzzy":2},"total_outbound_references":57},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 57 of 57 outbound references and 0 inbound Pith citation observations for arXiv:2505.16972."}