{"as_of":"2026-08-10T14:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ede944a495ac690a321ba27a3bdc04a8466ef9945cde9d6a972713e8e192e4c3","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:22:31.869618Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T15:52:44.054140Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T17:07:12.867079Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19206","snapshot_observed_at":"2026-08-05T15:52:44.054140Z","title":"Speak- stream: Streaming text-to-speech with interleaved data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.00078","last_updated":"2025-08-26T20:40:24Z","snapshot_observed_at":"2026-08-05T15:52:43.760838Z","submitted_at":"2025-08-26T20:40:24Z","title":"ChipChat: Low-Latency Cascaded Conversational Agent in MLX","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T15:52:44.054140Z"},"links":{"cited_paper":"/paper/2505.19206","citing_paper":"/paper/2509.00078"},"observation_digest":"sha256:7b6061dc79c41e02d8785aa40de5cf10100a530bf77e40a087af597bcaff1e66","observation_id":"5fea5dc5-c540-4237-bc1d-d1f003ccf490","resolution":{"observed_at":"2026-08-05T15:52:44.054140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"cited_work":{"arxiv_id":"2505.19206","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19206","snapshot_observed_at":"2026-07-02T17:07:12.867079Z","title":"Speakstream: Streaming text-to-speech with interleaved data,","venue":null,"work_id":"a2f61bd2-d49f-4e3f-b19c-11b4f1d488af","year":2025},"citing_paper":{"arxiv_id":"2606.06991","last_updated":"2026-06-05T07:29:20Z","snapshot_observed_at":"2026-08-10T11:26:35.969932Z","submitted_at":"2026-06-05T07:29:20Z","title":"Don't Pause: Streaming Video-Language Synchrony for Online Video Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T22:11:01.690237Z"},"links":{"cited_paper":"/paper/2505.19206","citing_paper":"/paper/2606.06991"},"observation_digest":"sha256:87f2aa7d0c6f78aeef6e0cfe83c6b7e802223f6679792e46be8aadb664e963b2","observation_id":"6227ce77-3ec2-40de-9470-32c1250b84ee","resolution":{"observed_at":"2026-07-02T17:07:12.868675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.19206/citation-record","integrity":"/paper/2505.19206/integrity","json":"/paper/2505.19206/citation-record.json","paper":"/paper/2505.19206"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.538315Z","title":"Audi- olm: a language modeling approach to audio generation,","venue":null,"work_id":"8a70d7ca-c712-4ec9-869f-1aa34b05dafc","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.401774Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:d0b37deb1a6f703e91f592fd6208c57f2e05db221ddfeb7d4e00cf14aabc20f6","observation_id":"5c797376-1075-4f7a-b805-ed8753374b7c","resolution":{"observed_at":"2026-08-07T14:22:33.551223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20215","last_updated":"2025-03-26T04:17:55Z","snapshot_observed_at":"2026-08-06T08:46:20.194739Z","submitted_at":"2025-03-26T04:17:55Z","title":"Qwen2.5-Omni Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20215","snapshot_observed_at":"2026-08-07T14:22:30.544240Z","title":"Qwen2.5-omni technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.544240Z"},"links":{"cited_paper":"/paper/2503.20215","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:fb6e0d77f30a31bd04bfe1bfe65ddddfd043a9d136ecb4688a2d9ae65fca349d","observation_id":"2a49efeb-c160-4df2-b2ed-2eb489baef6e","resolution":{"observed_at":"2026-08-07T14:22:30.544240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:30.595293Z","title":"Spirit-lm: Interleaved spoken and written language model,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.595293Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:eece483a4405d159f114efc41b439b00884cb97cf6154bc601f1a8ef73df3309","observation_id":"bc980c59-7ca6-4cfc-9ad1-f31144368a47","resolution":{"observed_at":"2026-08-07T14:22:30.595293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19168","last_updated":"2024-10-24T21:20:10Z","snapshot_observed_at":"2026-07-06T19:39:23.839070Z","submitted_at":"2024-10-24T21:20:10Z","title":"MMAU: A Massive Multi-Task Audio Understanding and Reasoning Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19168","snapshot_observed_at":"2026-08-07T14:22:30.673231Z","title":"Mmau: A massive multi- task audio understanding and reasoning benchmark,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.673231Z"},"links":{"cited_paper":"/paper/2410.19168","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:7c3bc499c521c24dd42690e9a0cf0b6554b6db5ade3347714848641c911bb092","observation_id":"c1bc2262-c193-473e-821f-5069f1f8eb6d","resolution":{"observed_at":"2026-08-07T14:22:30.673231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00767","last_updated":"2024-10-01T15:04:21Z","snapshot_observed_at":"2026-08-08T20:44:37.961271Z","submitted_at":"2024-10-01T15:04:21Z","title":"Zero-Shot Text-to-Speech from Continuous Text Streams","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00767","snapshot_observed_at":"2026-08-07T14:22:30.726922Z","title":"Zero- shot text-to-speech from continuous text streams,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.726922Z"},"links":{"cited_paper":"/paper/2410.00767","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:b8851db5f934c8aca80ca9b803bd480d6e3f1cb80e1feb761331ac4dff0e91ae","observation_id":"0b239529-5d94-462f-9b54-9ad986574e10","resolution":{"observed_at":"2026-08-07T14:22:30.726922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.467626Z","title":"Speak while you think: Streaming speech synthesis during text generation,","venue":null,"work_id":"7ffc38f1-f5bd-47dc-a8f2-373c1a11ea46","year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.780606Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:c95fd9abaec31558b61fba4d93e8b37dcfc523ee74b28cd0c09422ef98ca6025","observation_id":"8928467d-c253-4fba-ac9c-676f4539d528","resolution":{"observed_at":"2026-08-07T14:22:33.477254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16102","last_updated":"2025-08-09T10:01:51Z","snapshot_observed_at":"2026-08-10T08:02:08.842765Z","submitted_at":"2024-12-20T17:43:50Z","title":"Interleaved Speech-Text Language Models for Simple Streaming Text-to-Speech Synthesis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16102","snapshot_observed_at":"2026-08-07T14:22:30.842860Z","title":"Interleaved speech-text language models are simple streaming text to speech synthesizers,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.842860Z"},"links":{"cited_paper":"/paper/2412.16102","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:8102e53f98698eba06f0bf95215f9727ccb9616ca12686c32b81b2b8070041c9","observation_id":"b18d51a4-4449-433f-b0d8-7f587de7c478","resolution":{"observed_at":"2026-08-07T14:22:30.842860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.15835","last_updated":"2025-05-21T16:55:34Z","snapshot_observed_at":"2026-08-09T20:49:08.617871Z","submitted_at":"2024-07-22T17:51:53Z","title":"dMel: Speech Tokenization made Simple","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.15835","snapshot_observed_at":"2026-08-07T14:22:30.915758Z","title":"dmel: Speech tokenization made simple,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:30.915758Z"},"links":{"cited_paper":"/paper/2407.15835","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:b2575b324135d622026151b31cc9495d0dd8fac088d024f5060c16e2f325ce69","observation_id":"c136ad76-0c3f-424d-97f2-a7621681b0e8","resolution":{"observed_at":"2026-08-07T14:22:30.915758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.412663Z","title":"A 3T: Alignment-aware acoustic and text pretraining for speech synthesis and editing,","venue":null,"work_id":"6b55e55b-a3a1-497b-b96b-6d4d6d841464","year":2022},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.004924Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:f29f0dee7789e7cf4a7a3377e72266cc9525ccd56cd9652681dc09019f9ed545","observation_id":"4d648051-cf1a-46e1-9b8e-67188104c6c8","resolution":{"observed_at":"2026-08-07T14:22:33.434512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.029594Z","title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.029594Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:798b64aec8b45fabaa842123e8968b5d5678a5723b1cfe7285b8d5c0a3674522","observation_id":"9ffb4b08-22c6-47bd-ad27-01f64133f11b","resolution":{"observed_at":"2026-08-07T14:22:31.029594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T14:22:31.087888Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.087888Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:ddee176f1e5b39e22131ae2a2e3efb64bac3e163d8af2477f7e4476218a1930e","observation_id":"204d923a-49c9-443c-ad21-c47bea9e7cbb","resolution":{"observed_at":"2026-08-07T14:22:31.087888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.331296Z","title":"E3 tts: Easy end-to- end diffusion-based text to speech,","venue":null,"work_id":"291e054f-8736-42a8-9191-bea4ff8afd5f","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.132497Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:6d35366d98f80072ea7b60b8b47b8193089a1bc979218ec55ea3901764f4eca4","observation_id":"69c1a5bb-5c73-4bd8-8a8f-4cba12c0e441","resolution":{"observed_at":"2026-08-07T14:22:33.343081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-06T18:05:37.673476Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-07T14:22:31.169600Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.169600Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:817923dcd75bb7c1669706fdfc0658c0772d1e9028c347b699d9de4f2540e39f","observation_id":"7fc88f8e-e561-4c13-89a6-35e5a654f82f","resolution":{"observed_at":"2026-08-07T14:22:31.169600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1703.10135","last_updated":"2017-04-06T21:20:34Z","snapshot_observed_at":"2026-08-05T15:51:16.043022Z","submitted_at":"2017-03-29T16:55:13Z","title":"Tacotron: Towards End-to-End Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1703.10135","snapshot_observed_at":"2026-08-07T14:22:31.221168Z","title":"Tacotron: Towards end- to-end speech synthesis,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.221168Z"},"links":{"cited_paper":"/paper/1703.10135","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:0ecb07f7aa6523b56ba581447cf809b6980040a8b9fb73404d62b770dd2a833d","observation_id":"6fa8f085-6f5f-4e23-8147-693b3da14f4e","resolution":{"observed_at":"2026-08-07T14:22:31.221168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.289041Z","title":"(2024) Text-to-speech guide","venue":null,"work_id":"ea25e403-99a4-4c90-8181-063317ecf1bc","year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.249785Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:7967abfd9067a99b22a9095c4a0066c814168c439b2f2dd77ee6d7c936494e81","observation_id":"30b3b16d-7403-4add-a2d3-9a216fec6c90","resolution":{"observed_at":"2026-08-07T14:22:33.304761Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.242946Z","title":"Streamspeech: Low-latency neural architecture for high-quality on-device speech synthesis,","venue":null,"work_id":"c8ce0572-7689-4edd-a571-359491390d08","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.282325Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:b8cc0dd945f25ea30e21a583d36f0991a588f2a6636eb7d75698a84cde4079b8","observation_id":"1efec5e5-675f-4cf5-8bed-cd3d766ca544","resolution":{"observed_at":"2026-08-07T14:22:33.252460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.327328Z","title":"MLX: Efficient and flexible machine learning on apple silicon,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.327328Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:b159536ac5128dfd79e8488f4ad8d92552e61daedc7c61df8abe8047897e4a66","observation_id":"6209cf8c-2fd9-47ab-b891-3aee83ad4a78","resolution":{"observed_at":"2026-08-07T14:22:31.327328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14321","last_updated":"2025-03-14T00:10:58Z","snapshot_observed_at":"2026-08-09T13:51:45.924769Z","submitted_at":"2024-01-25T17:19:01Z","title":"VALL-T: Decoder-Only Generative Transducer for Robust and Decoding-Controllable Text-to-Speech","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14321","snapshot_observed_at":"2026-08-07T14:22:31.416374Z","title":"Vall-t: Decoder-only generative transducer for robust and decoding-controllable text-to-speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.416374Z"},"links":{"cited_paper":"/paper/2401.14321","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:8e75213aa7b792ce70800e00004888d4c1eb3dee2bcd80b0933eb11a61502412","observation_id":"e261da5d-e116-4fe1-b3ab-5a9103f733a5","resolution":{"observed_at":"2026-08-07T14:22:31.416374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.11943","last_updated":"2021-03-22T15:34:39Z","snapshot_observed_at":"2026-08-10T10:17:45.578054Z","submitted_at":"2021-03-22T15:34:39Z","title":"BERT: A Review of Applications in Natural Language Processing and Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.11943","snapshot_observed_at":"2026-08-07T14:22:31.501147Z","title":"Bert: a review of applications in natural language processing and understanding,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.501147Z"},"links":{"cited_paper":"/paper/2103.11943","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:252028fdf6f42af7bca98a0baec032639e8f3cb4ad044e14ca4642ceaf9fba38","observation_id":"20ab1739-357b-457d-91d3-b16a217f4b12","resolution":{"observed_at":"2026-08-07T14:22:31.501147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.04301","last_updated":"2021-05-11T04:12:14Z","snapshot_observed_at":"2026-08-06T20:00:38.096518Z","submitted_at":"2020-10-08T23:41:39Z","title":"Non-Attentive Tacotron: Robust and Controllable Neural TTS Synthesis Including Unsupervised Duration Modeling","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.04301","snapshot_observed_at":"2026-08-07T14:22:31.555872Z","title":"Non-attentive tacotron: Robust and controllable neural tts synthesis including unsupervised duration modeling,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.555872Z"},"links":{"cited_paper":"/paper/2010.04301","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:ee1a3e2e07f3d7401adce8d6ff96f4f30bfa559f4b79e37c8f35c06a47e7339b","observation_id":"d9bcda6d-b7a2-4e38-90e2-dbf56765bdbf","resolution":{"observed_at":"2026-08-07T14:22:31.555872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02897","last_updated":"2024-06-10T05:50:53Z","snapshot_observed_at":"2026-07-31T21:41:43.347268Z","submitted_at":"2024-06-05T03:36:11Z","title":"LiveSpeech: Low-Latency Zero-shot Text-to-Speech via Autoregressive Modeling of Audio Discrete Codes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02897","snapshot_observed_at":"2026-08-07T14:22:31.639465Z","title":"Livespeech: Low- latency zero-shot text-to-speech via autoregressive modeling of audio discrete codes,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.639465Z"},"links":{"cited_paper":"/paper/2406.02897","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:326b2ae53ed319375dacec23228beac1025ee6de2193f1f022193132e6b1fe78","observation_id":"264a593b-6af6-4e94-a1a6-e63767c50fec","resolution":{"observed_at":"2026-08-07T14:22:31.639465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.168743Z","title":"Transduce and speak: Neural transducer for text-to-speech with semantic token predic- tion,","venue":null,"work_id":"28cf1a42-1932-4649-913d-9fa4524943ef","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.674377Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:7dc21cfa8fb19a92997f3012dd5d3aca78d24cae0248e688db304ad33af08f36","observation_id":"b70e1eb8-b561-42d4-bcc2-511129d73975","resolution":{"observed_at":"2026-08-07T14:22:33.190432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.715642Z","title":"Parallel wavegan: A fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.715642Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:f23059c00e0d2f8ea22c9938f69d55ec75dd900704e305cbe82b3778922cd354","observation_id":"fcf3c1f7-9d52-4c8b-b2ef-cf855d75e533","resolution":{"observed_at":"2026-08-07T14:22:31.715642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.733397Z","title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.733397Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:3349b93a475de3950d560618f8b1939b46d999665abe8033f911f5d5f0ab72cd","observation_id":"5fc08e52-2a4b-4dfc-88ac-3e787debc790","resolution":{"observed_at":"2026-08-07T14:22:31.733397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:33.047999Z","title":"Bigvgan: A universal neural vocoder with large-scale training,","venue":null,"work_id":"84c4ffe9-7ce0-4c81-88fa-64bdcb681be4","year":null},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.746547Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:e6f0575c224ac917a637aecb9f70994b6f8a5a3ce2d0c92902a231ce177aadd8","observation_id":"bf912353-3894-4d2b-86e5-1a360d45569b","resolution":{"observed_at":"2026-08-07T14:22:33.075120Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.985422Z","title":"V ocos: Closing the gap between time-domain and fourier- based neural vocoders for high-quality audio synthesis,","venue":null,"work_id":"0ba34996-4e4a-47b0-bbb9-a5be154e0e50","year":null},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.759386Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:ffcb0f9ef24ff113a15c92805b854c3d3c63ca9a9434a4321f9662719f4230b2","observation_id":"939b6549-48d9-4fc6-b00e-96a60271dc9c","resolution":{"observed_at":"2026-08-07T14:22:33.020489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.935198Z","title":"Non-causal to causal ssl-supported transfer learning: Towards a high-performance low-latency speech vocoder,","venue":null,"work_id":"e1273222-3360-43ea-b973-d3abb214957f","year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.778855Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:88d51755effcf2166065b688885d90658b580dbf18889f950fb9df9af35199d5","observation_id":"027f9df9-6925-4d0d-941b-fc979cc4d52a","resolution":{"observed_at":"2026-08-07T14:22:32.945758Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.892301Z","title":"Coqui TTS: A deep learning toolkit for Text-to-Speech, battle-tested in research and production,","venue":null,"work_id":"358e83f7-7c15-403d-b5f2-a3d0edad532c","year":null},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.787676Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:d6221512da5c7142f921b43ec069b7eefcd737ac190a3ab1135f9481d456cf2f","observation_id":"64393298-7ab5-4cbc-933d-fc9615b7d845","resolution":{"observed_at":"2026-08-07T14:22:32.906779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.807757Z","title":"The lj speech dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.807757Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:3d6b54f2bfe30a8146cdf6c6aacaf304d5ba405ac09ef352cb805e3a490541be","observation_id":"71b7ac5a-074f-4634-8442-ef81b3ea1d29","resolution":{"observed_at":"2026-08-07T14:22:31.807757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.748753Z","title":"Whisperx: Time-accurate speech transcription of long-form audio,","venue":null,"work_id":"67023347-daf6-41e5-ab6e-205d9831c33d","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.816394Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:9ec74f2c2b13642346337b039c3ef49742f27a34d090c5eeac47027dbb8739f6","observation_id":"3fd2acba-0c0e-4856-a3d9-9b89e5eb8477","resolution":{"observed_at":"2026-08-07T14:22:32.766258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.823985Z","title":"Robust speech recognition via large-scale weak super- vision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.823985Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:1397e4d93f7867bc42cfec7a3c1d15f7192f9e31bbfdf2d28d67eab48cb829cc","observation_id":"ad6a3e0d-0db1-425d-96dd-3d850d4678a6","resolution":{"observed_at":"2026-08-07T14:22:31.823985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.661703Z","title":"Libritts-r: A restored multi-speaker text-to-speech corpus,","venue":null,"work_id":"ed81fdcd-bf47-435c-ab75-8e664fb456f4","year":2023},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.845205Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:392d5201600b1bab16b4070640c131c6f9962f6b35a14b172f0bf4b7041ee0c4","observation_id":"44a264d3-1988-48c0-bc65-f38876f91a97","resolution":{"observed_at":"2026-08-07T14:22:32.674771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:31.853069Z","title":"Librispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.853069Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:515bcda716836159a9d360e47a37fb099024b6e3054470a73645f615a8e7237e","observation_id":"f1fbf252-58e1-4706-91c0-32a9c352fb57","resolution":{"observed_at":"2026-08-07T14:22:31.853069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-07T14:22:31.869618Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.869618Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:86b04f72eb70f9b436d9a8e949a725118220d5fa8c4465092e1bb3ba095f0126","observation_id":"841334ad-98d8-4578-a9aa-8572b38c2b0b","resolution":{"observed_at":"2026-08-07T14:22:31.869618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:22:32.840509Z","title":"Available: https://www.coqui.ai","venue":null,"work_id":"7fb60e6d-525c-4fc1-9819-891055cb2cfe","year":null},"citing_paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-07T14:22:31.800042Z"},"links":{"citing_paper":"/paper/2505.19206"},"observation_digest":"sha256:b161c7e5abe86c62f1b94f0bcb9dc6cd91b1be33e06b67957c8e1766eceeeb38","observation_id":"d2eae670-95c0-493b-8556-5d1d38b0af9b","resolution":{"observed_at":"2026-08-07T14:22:32.851116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.19206","last_updated":"2025-05-25T16:11:10Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T12:05:27.001330Z","submitted_at":"2025-05-25T16:11:10Z","title":"SpeakStream: Streaming Text-to-Speech with Interleaved Data"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 2 inbound Pith citation observations for arXiv:2505.19206."}