{"as_of":"2026-08-10T12:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:53b5f155ff5ff4a1c9ba1050b3803c2b2b5bfcbbd88d57fd6964a317c542c9bf","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":19,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":19,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":19,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:23:40.366858Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T14:38:28.620667Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2505.17589","last_updated":"2025-05-27T07:48:34Z","snapshot_observed_at":"2026-08-08T16:08:45.949013Z","submitted_at":"2025-05-23T07:55:21Z","title":"CosyVoice 3: Towards In-the-wild Speech Generation via Scaling-up and Post-training","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-16T05:27:25.425188Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.17589"},"observation_digest":"sha256:48569223e7e749b1eb9a748ce7600fd29cdf15feb397929f9ed93ba3d6b5fbc2","observation_id":"0d5f7645-2d04-4d74-be77-c2ff20279371","resolution":{"observed_at":"2026-05-16T05:27:25.612037Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-07T14:23:40.366858Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19103","last_updated":"2025-05-25T11:45:08Z","snapshot_observed_at":"2026-08-09T09:56:31.596522Z","submitted_at":"2025-05-25T11:45:08Z","title":"WHISTRESS: Enriching Transcriptions with Sentence Stress Detection","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:23:40.366858Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.19103"},"observation_digest":"sha256:74341e51e2207430b5a9b974fa0ac4087ba87be70bd1b640a4bb7f72bcbdc5ac","observation_id":"742884e6-2cf3-47cd-8ac0-ae6d3c53620f","resolution":{"observed_at":"2026-08-07T14:23:40.366858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2505.22765","last_updated":"2026-04-07T15:07:17Z","snapshot_observed_at":"2026-07-06T21:32:28.358564Z","submitted_at":"2025-05-28T18:32:56Z","title":"StressTest: Can YOUR Speech LM Handle the Stress?","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-19T13:00:23.002962Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2505.22765"},"observation_digest":"sha256:6f28f80a788670ca36e5f00d34a82d5f160d6a53154e75978d7a42226ec9a9ea","observation_id":"5ab751e6-8225-4415-82ca-520f8cb83fcc","resolution":{"observed_at":"2026-05-19T13:02:18.259217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-07T10:55:20.345371Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.04013","last_updated":"2025-06-04T14:42:12Z","snapshot_observed_at":"2026-08-08T13:06:14.713555Z","submitted_at":"2025-06-04T14:42:12Z","title":"Towards Better Disentanglement in Non-Autoregressive Zero-Shot Expressive Voice Conversion","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T10:55:20.345371Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2506.04013"},"observation_digest":"sha256:8e929c9fa1010c9ea44638b9bbf7d096f32e67e73f50fea976d3cdad1416e4c7","observation_id":"a5e2b151-c0e1-4e2c-beb8-922275a56e54","resolution":{"observed_at":"2026-08-07T10:55:20.345371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T16:33:52.438991Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13155","last_updated":"2025-07-17T14:17:40Z","snapshot_observed_at":"2026-08-07T19:53:51.352286Z","submitted_at":"2025-07-17T14:17:40Z","title":"NonverbalTTS: A Public English Corpus of Text-Aligned Nonverbal Vocalizations with Emotion Annotations for Text-to-Speech","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:52.438991Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2507.13155"},"observation_digest":"sha256:da9e11b728587ea20314a4e17452c674a5247b4a399956232259809be2f5d3a8","observation_id":"79602e4a-bfe0-4eec-9975-1136cfc0f26c","resolution":{"observed_at":"2026-08-06T16:33:52.438991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T00:51:31.393989Z","title":"A.; Hsu, W.-N.; d'Avirro, A.; Shi, B.; Gat, I.; Fazel-Zarani, M.; Remez, T.; Copet, J.; Synnaeve, G.; Hassid, M.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.04195","last_updated":"2025-08-06T08:25:26Z","snapshot_observed_at":"2026-08-06T17:34:54.346667Z","submitted_at":"2025-08-06T08:25:26Z","title":"NVSpeech: An Integrated and Scalable Pipeline for Human-Like Speech Modeling with Paralinguistic Vocalizations","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T00:51:31.393989Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2508.04195"},"observation_digest":"sha256:f5bf970dc8302c1fb1d7b1951d3f04a4a3f6c5afb08bef1e869f8db78e00bab5","observation_id":"4402f487-40f7-461f-be2d-02e40eaa173d","resolution":{"observed_at":"2026-08-06T00:51:31.393989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2509.04072","last_updated":"2026-04-21T15:32:24Z","snapshot_observed_at":"2026-07-06T22:23:48.679487Z","submitted_at":"2025-09-04T10:05:06Z","title":"Computational Narrative Understanding for Expressive Text-to-Speech","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T19:30:53.250247Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2509.04072"},"observation_digest":"sha256:696408cee8006724bc8f38ed8ccd8a6fd10da84ee9fb26b9258d5a275a4dc024","observation_id":"2bbcfe48-1e89-402a-ad67-8a90aa46b3da","resolution":{"observed_at":"2026-05-18T19:31:47.053839Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2509.22220","last_updated":"2026-04-13T11:56:11Z","snapshot_observed_at":"2026-08-02T12:47:50.534468Z","submitted_at":"2025-09-26T11:32:51Z","title":"StableToken: A Noise-Robust Semantic Speech Tokenizer for Resilient SpeechLLMs","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-18T12:57:04.450462Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2509.22220"},"observation_digest":"sha256:8ab636f9ea9dcb48b692991df714cabc7ae3f54f4544acfd47cf306153bfeefa","observation_id":"8a8b3a10-c5d3-42aa-bd5c-3168ab4bdcfe","resolution":{"observed_at":"2026-05-18T13:01:24.344834Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-03T04:28:01.655676Z","title":"InFindings of the Association for Com- putational Linguistics: NAACL 2025, pages 1604– 1635, Albuquerque, New Mexico","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05027","last_updated":"2026-02-06T13:35:19Z","snapshot_observed_at":"2026-08-03T12:01:43.618607Z","submitted_at":"2026-02-04T20:29:16Z","title":"AudioSAE: Towards Understanding of Audio-Processing Models with Sparse AutoEncoders","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T04:28:01.655676Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2602.05027"},"observation_digest":"sha256:478a2c6a2615b31c5d12061e2cbce4de6dacb9402c64a1138f8b7c3fee29e1e3","observation_id":"0252de93-6edd-4e86-b63c-26a9a7e93e21","resolution":{"observed_at":"2026-08-03T04:28:01.655676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2603.25551","last_updated":"2026-04-06T15:20:14Z","snapshot_observed_at":"2026-07-06T22:50:41.736941Z","submitted_at":"2026-03-26T15:23:34Z","title":"Voxtral TTS","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T00:38:42.441340Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2603.25551"},"observation_digest":"sha256:ef67de04a7d81c8d68ded834f18d0bedd81584eee9ced985fbe0166c25f1bbba","observation_id":"d8ea44ac-40cb-4828-8173-552871608972","resolution":{"observed_at":"2026-05-15T00:39:35.896911Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.04505","last_updated":"2026-05-06T05:18:42Z","snapshot_observed_at":"2026-08-02T05:45:46.058583Z","submitted_at":"2026-05-06T05:18:42Z","title":"JASTIN: Aligning LLMs for Zero-Shot Audio and Speech Evaluation via Natural Language Instructions","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-08T16:43:33.397158Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.04505"},"observation_digest":"sha256:4aed9128fbbe83f8d7fc4b284acfd7ef9d5f825fb197ced47821624cab3b68fb","observation_id":"549b17d1-640d-431b-8730-8f5d3ea72df8","resolution":{"observed_at":"2026-05-11T18:01:08.380039Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.20830","last_updated":"2026-05-20T07:21:36Z","snapshot_observed_at":"2026-08-02T05:58:18.404966Z","submitted_at":"2026-05-20T07:21:36Z","title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T02:32:26.122526Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.20830"},"observation_digest":"sha256:f540067236365b09a6b932ca11897cfb33bf8cf00c9c10f2e8cfd3f7c862a1c0","observation_id":"0661bbb5-e44f-4264-8a08-e077c33939ac","resolution":{"observed_at":"2026-05-21T02:33:55.471353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.25504","last_updated":"2026-05-25T07:08:58Z","snapshot_observed_at":"2026-08-02T14:20:09.406007Z","submitted_at":"2026-05-25T07:08:58Z","title":"Toward Natural Emotional Text-To-Speech System with Fine-Grained Non-Verbal Expression Control","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-29T20:53:47.252916Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.25504"},"observation_digest":"sha256:05d7fb8596ad9031d9777d74c7fa53f39016830849a036de01c0cb83c69ad17e","observation_id":"ff5f36a2-5395-4982-b1d5-1428c3a7f1f3","resolution":{"observed_at":"2026-06-29T20:53:57.753462Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2605.30748","last_updated":"2026-06-01T01:53:31Z","snapshot_observed_at":"2026-07-06T23:39:58.378823Z","submitted_at":"2026-05-29T02:25:02Z","title":"Chatterbox-Flash: Prior-Calibrated Block Diffusion for Streaming Zero-Shot TTS","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T21:31:09.399885Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2605.30748"},"observation_digest":"sha256:3943e292f1b0c1c045683b1c1b82570a4880fe4bd1d601c80bacf11ff1b8884d","observation_id":"b73b78d3-4db9-431a-b7d5-8d14dca35cc9","resolution":{"observed_at":"2026-07-01T20:16:11.068632Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2606.01677","last_updated":"2026-06-01T04:35:28Z","snapshot_observed_at":"2026-08-01T15:52:02.822088Z","submitted_at":"2026-06-01T04:35:28Z","title":"UniVocal: Unified Speech-Singing Code-Switching Synthesis","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-06-28T13:17:13.510587Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2606.01677"},"observation_digest":"sha256:e0624fd4067c656e268949f1fbbb2d6d0a748306e9c8201c1e8a6fc3170176af","observation_id":"0ef35b23-59a6-4ae5-b80e-62ec4283707a","resolution":{"observed_at":"2026-07-02T00:46:24.556094Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":"2308.05725","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-07-03T14:38:28.620667Z","title":"[TARGET]","venue":null,"work_id":"0ae887a4-964c-4515-8a7e-7e4fc1e07dd1","year":2026},"citing_paper":{"arxiv_id":"2607.02214","last_updated":"2026-07-02T14:22:46Z","snapshot_observed_at":"2026-07-07T00:07:42.752664Z","submitted_at":"2026-07-02T14:22:46Z","title":"Unlocking Speech-Text Compositional Powers: Instruction-Following Speech Language Models without Instruction Tuning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-03T14:35:17.004680Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.02214"},"observation_digest":"sha256:c402cb28f264ad1ff64f3216d1158fee8e54ced43ed1092047b578010856fb1e","observation_id":"483f3ab7-4cda-43af-8e3c-9493413f6c44","resolution":{"observed_at":"2026-07-03T14:38:28.622473Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-02T02:30:34.179334Z","title":"Expresso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.14310","last_updated":"2026-07-15T19:16:31Z","snapshot_observed_at":"2026-08-09T15:59:59.649408Z","submitted_at":"2026-07-15T19:16:31Z","title":"Dialogs: a studio-quality expressive conversational Russian speech corpus for dialog assistants","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T02:30:34.179334Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.14310"},"observation_digest":"sha256:53623020ffb0f00ee37344d0a31de7435bdc3a9993d6f3e41e0af35400b61b45","observation_id":"612af17a-827e-4187-b885-0b712fcafe7d","resolution":{"observed_at":"2026-08-02T02:30:34.179334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-01T07:12:10.474782Z","title":"arXiv preprint arXiv:2308.05725 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21550","last_updated":"2026-07-23T17:35:20Z","snapshot_observed_at":"2026-08-08T08:48:36.880078Z","submitted_at":"2026-07-23T17:35:20Z","title":"X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-01T07:12:10.474782Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2607.21550"},"observation_digest":"sha256:c1c9dfd24c5c1b23b8dab510227af42258506fc98a9ba2b98bd1337c23510623","observation_id":"256e3517-1a63-4ca0-b573-407c72110a94","resolution":{"observed_at":"2026-08-01T07:12:10.474782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05725","snapshot_observed_at":"2026-08-06T00:40:33.274250Z","title":"Ex- presso: A benchmark and analysis of discrete expressive speech resynthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00998","last_updated":"2026-08-02T04:54:12Z","snapshot_observed_at":"2026-08-09T16:27:06.444887Z","submitted_at":"2026-08-02T04:54:12Z","title":"Beyond One-Size-Fits-All: Personalized and Culturally Adaptive Emotional TTS via Interactive Optimization of Individual Emotion Perception Spaces","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T00:40:33.274250Z"},"links":{"cited_paper":"/paper/2308.05725","citing_paper":"/paper/2608.00998"},"observation_digest":"sha256:f2cd6374bf12b92c731c732b1cd3c47b3cf965e0555f02cf9d75dd4b8f887ce4","observation_id":"fec8ed68-93ac-427b-a97e-d3dea908ede9","resolution":{"observed_at":"2026-08-06T00:40:33.274250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2308.05725/citation-record","integrity":"/paper/2308.05725/integrity","json":"/paper/2308.05725/citation-record.json","paper":"/paper/2308.05725"},"outbound":[],"paper":{"arxiv_id":"2308.05725","last_updated":"2023-08-10T17:41:19Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T17:41:19Z","title":"EXPRESSO: A Benchmark and Analysis of Discrete Expressive Speech Resynthesis"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 19 inbound Pith citation observations for arXiv:2308.05725."}