{"as_of":"2026-08-10T17:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d6841a47a8b5632136998b6f2390141ab26aa9dc5130f0ff96a67de088c1c0f9","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:50:17.390376Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T20:50:17.213254Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-09T20:50:17.551912Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"cited_work":{"arxiv_id":"2501.19258","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.19258","snapshot_observed_at":"2026-08-09T20:50:17.551912Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","venue":"cs.CL","work_id":"f37b8878-1039-45e9-8df9-3ba9e0feea6f","year":2025},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.213254Z"},"links":{"cited_paper":"/paper/2501.19258","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:e93ba54edc421897ead3953a425c921e1eb3f7529bb2bfb1a0fcd755b757c20b","observation_id":"46264d7c-d2c7-418f-8164-aa4dcda14f9b","resolution":{"observed_at":"2026-08-09T20:50:17.559362Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2501.19258/citation-record","integrity":"/paper/2501.19258/integrity","json":"/paper/2501.19258/citation-record.json","paper":"/paper/2501.19258"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.989705Z","title":"Despite the high-quality output, the prosody of the generated speech sometimes be- comes inappropriate for the context","venue":null,"work_id":"13c7b242-499a-4a16-bd65-a560e469666f","year":null},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.207333Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:4a2352efab8daeeb60e764053ec19608be455f6daf33bc8b6cb65a3400584968","observation_id":"5a1cde93-cc78-44a7-a999-ef99aa50718d","resolution":{"observed_at":"2026-08-09T20:50:17.994955Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"cited_work":{"arxiv_id":"2501.19258","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.19258","snapshot_observed_at":"2026-08-09T20:50:17.551912Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","venue":"cs.CL","work_id":"f37b8878-1039-45e9-8df9-3ba9e0feea6f","year":2025},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.213254Z"},"links":{"cited_paper":"/paper/2501.19258","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:e93ba54edc421897ead3953a425c921e1eb3f7529bb2bfb1a0fcd755b757c20b","observation_id":"46264d7c-d2c7-418f-8164-aa4dcda14f9b","resolution":{"observed_at":"2026-08-09T20:50:17.559362Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.973399Z","title":"Experimental Setup To assess the impact of visual information on speech generation performance, a dataset containing both diverse prosodic and vi- sual data is essential","venue":null,"work_id":"2efe053a-513e-47d0-9094-41de145fe96b","year":null},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.218991Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:0d19f1a2b664ebf31872a48f8bab43cfe33407d1f33de9adb1245b96eaae971f","observation_id":"614772b2-901a-40fa-901d-0bc9060921b3","resolution":{"observed_at":"2026-08-09T20:50:17.978937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.938182Z","title":"Using two dis- tinct video feature extractors, we demonstrate that these visual features encapsulate prosodic information","venue":null,"work_id":"cc5fc97b-c58c-4795-8b98-f28cfbd3b02c","year":null},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.229624Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:20518adf93a100436c15ee2f9643ed4c0822a12b6e886a0e8168edd2e8133f28","observation_id":"b65af53c-99b6-4490-8153-84449ea9e73f","resolution":{"observed_at":"2026-08-09T20:50:17.944262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.09934","last_updated":"2022-04-21T07:49:09Z","snapshot_observed_at":"2026-08-07T02:30:21.925645Z","submitted_at":"2022-04-21T07:49:09Z","title":"FastDiff: A Fast Conditional Diffusion Model for High-Quality Speech Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.09934","snapshot_observed_at":"2026-08-09T20:50:17.254499Z","title":"FastDiff: A fast conditional diffusion model for high-quality speech synthesis,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.254499Z"},"links":{"cited_paper":"/paper/2204.09934","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:a8e5a360d301c9b30d73d07624a7066fb7c9d04015756e5210918c59da600f7a","observation_id":"5116b954-f3ef-435a-bc36-ba1e01d24db1","resolution":{"observed_at":"2026-08-09T20:50:17.254499Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.919592Z","title":"Deep mixture density networks for acous- tic modeling in statistical parametric speech synthesis,","venue":null,"work_id":"a7c5904a-d569-48d3-8216-6beaad87f7c4","year":2014},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.234534Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:ba926481f090d2d7fc07b698f3eb9f8a2f353d940a82fabd75ba8833840e8263","observation_id":"9638055a-4317-40f2-8b01-ae94bdd9e1ea","resolution":{"observed_at":"2026-08-09T20:50:17.925404Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.955763Z","title":null,"venue":null,"work_id":"e531f971-71b1-4809-a965-860dc1d71a78","year":null},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.224287Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:9f9263704ec16a313250ce67d127000e8ecbad1ef0cafa45440f8a4a90a9e660","observation_id":"6ef20369-d217-483d-b630-cd32c8fcf63c","resolution":{"observed_at":"2026-08-09T20:50:17.961376Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.900848Z","title":"Unidirectional long short-term memory re- current neural network with recurrent output layer for low-latency speech synthesis,","venue":null,"work_id":"0c8e5a79-ca7f-4fb5-9ebd-eb191808f938","year":2015},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.239311Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:66027f9051efb5d6e487b7a14159771bdf7068b5838239d49f11e72fd44c644d","observation_id":"88fe2c5f-9cc4-4a1b-b191-bdac925e8e76","resolution":{"observed_at":"2026-08-09T20:50:17.906544Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.883333Z","title":"NaturalSpeech: End-to-end text-to- speech synthesis with human-level quality,","venue":null,"work_id":"4f2c3538-858e-43d0-b0ea-726b9aecd3d8","year":2024},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.244348Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:1ee14d4ec537863c77df981865ed3b310480a6adde42b363ef233c4a829ecbf0","observation_id":"7bc7dcca-935c-4562-86a2-3d334cd4375a","resolution":{"observed_at":"2026-08-09T20:50:17.888796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.866028Z","title":"FastSpeech: Fast, robust and controllable text to speech,","venue":null,"work_id":"72a3e022-ae40-419d-9b24-126edb84317e","year":2019},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.249595Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:d4acfe6de007612d9cedd1268840984868c015e1c2b7f4a3655be5fcbb086f72","observation_id":"9d9b9273-93e7-4dd2-bef1-3602fcccba1b","resolution":{"observed_at":"2026-08-09T20:50:17.871652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.849257Z","title":"On granularity of prosodic representations in expressive text-to-speech,","venue":null,"work_id":"8d91f924-1d8c-460d-ab72-8f81eeb1426c","year":2022},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.260279Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:30dd11d289e533ea9edbb323c8b872d7d6956619b30c953296a58a7cfc53985c","observation_id":"6bf8f256-6251-4a71-921f-55f767396b09","resolution":{"observed_at":"2026-08-09T20:50:17.854616Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-06T18:05:37.673476Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-09T20:50:17.265052Z","title":"FastSpeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.265052Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:429713b7ec326278daefce352917e5fd1883c80f6ed8750235fdfec9cf5274c7","observation_id":"b3401b66-54ee-4800-a4b7-f3fed282a958","resolution":{"observed_at":"2026-08-09T20:50:17.265052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.832978Z","title":"Chive: Varying prosody in speech synthesis with a linguistically driven dynamic hierarchical conditional variational network,","venue":null,"work_id":"a118d46c-d2f4-4084-9c02-ff06adae820d","year":2019},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.275553Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:d37e04623f6d812248fd0e70b720409b72c8183e27fb0ee9111b2f9ee325b5b0","observation_id":"0c0bffc1-dbce-4135-85f1-968810ebaf38","resolution":{"observed_at":"2026-08-09T20:50:17.838085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.280420Z","title":"Mellotron: Mul- tispeaker expressive voice synthesis by conditioning on rhythm, pitch and global style tokens,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.280420Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:f192147bc5ecd0b3ac2673b28694fb76f4f6aaef9657c0c09288bf9d23e0067f","observation_id":"d0eea874-c8af-416c-b8dc-55541afae4b7","resolution":{"observed_at":"2026-08-09T20:50:17.280420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12708","last_updated":"2024-04-21T07:17:14Z","snapshot_observed_at":"2026-08-04T00:42:56.058282Z","submitted_at":"2023-05-22T04:37:41Z","title":"ViT-TTS: Visual Text-to-Speech with Scalable Diffusion Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12708","snapshot_observed_at":"2026-08-09T20:50:17.284979Z","title":"VIT-TTS: visual text-to-speech with scalable diffusion transformer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.284979Z"},"links":{"cited_paper":"/paper/2305.12708","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:52e5573d00008c7f28ea4b2f4f3c3cebc04b7365584c5fcde604ec78a133f13e","observation_id":"94ffe127-0a7d-4bff-af9a-4a3d08338dee","resolution":{"observed_at":"2026-08-09T20:50:17.284979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.289999Z","title":"The LJ Speech Dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.289999Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:b3ea75561c921a4ab168ed42bf52986120ab6691471dac210ec0cff0187e90b8","observation_id":"46b094d9-a8b9-4d94-819b-d3989ab742f9","resolution":{"observed_at":"2026-08-09T20:50:17.289999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02882","last_updated":"2019-04-05T06:05:00Z","snapshot_observed_at":"2026-08-10T09:39:47.608323Z","submitted_at":"2019-04-05T06:05:00Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.02882","snapshot_observed_at":"2026-08-09T20:50:17.294686Z","title":"LibriTTS: A corpus derived from librispeech for text- to-speech,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.294686Z"},"links":{"cited_paper":"/paper/1904.02882","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:ff54405b8ce35eb9774f6942bdcc62868014a80233078e260a2120fb3e3b4043","observation_id":"ed07cf1c-e7a6-4559-b713-9ea96e7bb626","resolution":{"observed_at":"2026-08-09T20:50:17.294686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.796049Z","title":"Ego4d: Around the world in 3,000 hours of egocentric video,","venue":null,"work_id":"b354ea15-088c-4a97-aae8-ab00facccd34","year":2022},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.300516Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:0b958d3b84a04ce86cea3e382fa818721d75df5f302f56db0cd459fde77d60b8","observation_id":"9a95dddd-a602-4907-a7dd-a1b6023ca39c","resolution":{"observed_at":"2026-08-09T20:50:17.801073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.780127Z","title":"Condensed movies: Story based retrieval with contextual embeddings,","venue":null,"work_id":"b5a2cf19-8654-498e-8231-487a1ae28728","year":2020},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.305703Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:bf734fef8bed4ee293daa6377b42b97c95eef722121a4add9277e183b54f070f","observation_id":"3fab26cc-05a9-4223-80fa-deed08cd241a","resolution":{"observed_at":"2026-08-09T20:50:17.784934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.764865Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":"fcd90520-9a1a-4fc3-bdc8-51aaa1f5d1fb","year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.310629Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:a47b6c17b9824d965350c2512d99e28ddc5876f2d19e0ca1a137527d57decd9c","observation_id":"c26d7058-5e60-4d59-8f14-ba95f6a3a9f1","resolution":{"observed_at":"2026-08-09T20:50:17.769915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.749489Z","title":"CMD+: A D.I.Y . Audiovisual Dataset for Multi- Speaker TTS,","venue":null,"work_id":"a2e59f9b-ab1f-4699-b086-ed4484e8d9aa","year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.316038Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:1137f5ef1f050855e67df0876c79a117f5abee94e7e1f938174e21d3a8d80eb4","observation_id":"f03ce37d-88da-4aec-a4cd-b3e2861d5cc6","resolution":{"observed_at":"2026-08-09T20:50:17.754642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.06979","last_updated":"2024-04-19T09:16:17Z","snapshot_observed_at":"2026-08-09T12:23:36.959806Z","submitted_at":"2023-08-14T07:32:03Z","title":"The Sound Demixing Challenge 2023 $\\unicode{x2013}$ Music Demixing Track","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.06979","snapshot_observed_at":"2026-08-09T20:50:17.320971Z","title":"The sound demixing challenge 2023 music demixing track,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.320971Z"},"links":{"cited_paper":"/paper/2308.06979","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:4958a46526fc376b05243a7c36038524ab4c61b9441eb15a5bc86b0604100ea4","observation_id":"ec659a45-2c03-4b10-bd2e-179f4d146a1d","resolution":{"observed_at":"2026-08-09T20:50:17.320971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.732988Z","title":"Resemblyzer,","venue":null,"work_id":"7de90749-4c4f-425a-9e2c-e8c633d94123","year":2024},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.326282Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:bd8c72a9a47680cc6b99e4ab87b028644a199ef38298f68a52a98c514b064a15","observation_id":"77c18112-6f8c-482e-b89b-e2c661ee6d81","resolution":{"observed_at":"2026-08-09T20:50:17.738232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15821","last_updated":"2023-12-25T22:24:49Z","snapshot_observed_at":"2026-08-05T03:40:24.447729Z","submitted_at":"2023-12-25T22:24:49Z","title":"Audiobox: Unified Audio Generation with Natural Language Prompts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15821","snapshot_observed_at":"2026-08-09T20:50:17.331087Z","title":"Audiobox: Unified au- dio generation with natural language prompts,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.331087Z"},"links":{"cited_paper":"/paper/2312.15821","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:5e99b97607ba5961d8cbe417bea41c3bfe29f071a17ae694e09be1af0fcf2155","observation_id":"ba086c40-95b3-425b-8253-8c5412fe4d0d","resolution":{"observed_at":"2026-08-09T20:50:17.331087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.716103Z","title":"A short- time objective intelligibility measure for time-frequency weighted noisy speech,","venue":null,"work_id":"ca4a9574-319a-4aeb-ae01-94561cbd304f","year":2010},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.336097Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:8c6189a8f120cddb2e263d4eba4debe9bcafe0c4c709b4e203e6b6eb834d80c3","observation_id":"d23f2617-43ca-43af-9b6f-d7c6a5031199","resolution":{"observed_at":"2026-08-09T20:50:17.721365Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.698844Z","title":"Per- ceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":"47d6c73f-b3e9-46ae-be9e-f9e0ff4725cf","year":2001},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.341247Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:0062649cb4c56d0132fbd95d22e432178f6ace7f14eb73cd4520870f1fe5486d","observation_id":"1fc38aa8-fb53-4dd2-9a6f-7a97c3cbc26a","resolution":{"observed_at":"2026-08-09T20:50:17.705037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.681721Z","title":"SDR– half-baked or well done?","venue":null,"work_id":"d8ba951a-f959-4133-88ec-08c94a028a5e","year":2019},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.346185Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:34ffc74f3cf838ff460598455264aac04a0d9990b10d7d46a11e2811d1595aea","observation_id":"8ac8f403-5036-45c5-b5a3-c09bb07886bd","resolution":{"observed_at":"2026-08-09T20:50:17.686765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.665203Z","title":"Torchaudio-Squim: Reference-less speech quality and intelligibility measures in torchaudio,","venue":null,"work_id":"687777f6-b9fe-444a-97a3-b64e8739f4e2","year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.350971Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:b9f9e29a24f25065737cd3b40abf75a3bfa3de9201e1ddc938360ce8f9927f60","observation_id":"0f45bc67-4145-4ae8-ac5e-f4f90fd1eec3","resolution":{"observed_at":"2026-08-09T20:50:17.670742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.646401Z","title":"Omnivore: A single model for many visual modalities,","venue":null,"work_id":"84a6b57b-db4c-4806-830d-824d5d55bb75","year":2022},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.356165Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:d9065355bd65c1f822688e94e4005bd93d7a978263c2561eb29d48e1a9092b6a","observation_id":"d918af71-c765-4722-8453-a3ce1a8793bc","resolution":{"observed_at":"2026-08-09T20:50:17.651564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.360860Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.360860Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:61607d09aa485d8a22ae5a297dbc6de29dab7e873fdd1832428d36ef57c6fb4f","observation_id":"a12931ea-8921-4708-afa1-3bc265c6a23e","resolution":{"observed_at":"2026-08-09T20:50:17.360860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6980","last_updated":"2017-01-30T01:27:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2014-12-22T13:54:29Z","title":"Adam: A Method for Stochastic Optimization","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1412.6980","snapshot_observed_at":"2026-08-09T20:50:17.365484Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.365484Z"},"links":{"cited_paper":"/paper/1412.6980","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:58d962cb013df4556d1497931194b7df6ee6e4066f8cb3079f30ec2880808866","observation_id":"de6c5db6-6bc8-4274-87e0-efa7ea411131","resolution":{"observed_at":"2026-08-09T20:50:17.365484Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.619705Z","title":"FastPitch: Parallel text-to-speech with pitch pre- diction,","venue":null,"work_id":"2f5f34db-97a6-4efe-b147-cd93f270d590","year":2021},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.370920Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:e2f7acb578758fdb6fcdcd6e212287b1225aa4219ab87c7afb3a5706bcdb7f88","observation_id":"dfe24b21-8c07-4d94-97e1-6b4b4d3a5c46","resolution":{"observed_at":"2026-08-09T20:50:17.624813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.603554Z","title":"Sonicvisionlm: Playing sound with vision language models,","venue":null,"work_id":"fd382db7-474c-4df9-b874-9fe6c9fdfa50","year":2024},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.375699Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:b98488d8aabf6c6a9e70a9edf79becb70c41ec7a9681bf03d49773fa432b37b3","observation_id":"e3e3b893-f7fc-4721-9143-ac3f359b7a27","resolution":{"observed_at":"2026-08-09T20:50:17.608653Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.588152Z","title":"End-to-end video-to-speech synthesis using gener- ative adversarial networks,","venue":null,"work_id":"e77dbcef-8fa6-4940-a04a-73e3d0dcc228","year":2022},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.380409Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:049604df7ab1f21cd52500d044c8cc810cdb7f60d4b147b3b017303d0e47547d","observation_id":"4eeae4f1-ef67-4de6-8687-25c0e93965a3","resolution":{"observed_at":"2026-08-09T20:50:17.593035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.19603","last_updated":"2023-05-31T07:17:32Z","snapshot_observed_at":"2026-07-06T15:35:49.019768Z","submitted_at":"2023-05-31T07:17:32Z","title":"Intelligible Lip-to-Speech Synthesis with Speech Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.19603","snapshot_observed_at":"2026-08-09T20:50:17.385174Z","title":"Intelligible lip-to-speech synthe- sis with speech units,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.385174Z"},"links":{"cited_paper":"/paper/2305.19603","citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:342c50f039e71e2dfce73cb002a0d1c58ebfd7ac5d006cfcdb8dc8689eb6534c","observation_id":"dce0b177-93b1-4bad-ac6b-1a923b5ebdff","resolution":{"observed_at":"2026-08-09T20:50:17.385174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T20:50:17.572403Z","title":"Camp: a two-stage approach to modelling prosody in context,","venue":null,"work_id":"1fb543ad-7ef5-43bb-9e88-e75c91b1ca9c","year":2021},"citing_paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-09T20:50:17.390376Z"},"links":{"citing_paper":"/paper/2501.19258"},"observation_digest":"sha256:cd9918fe0c28d8b2d355aa77b0a951de3d6c002541e099f57dae00d97fbcf00a","observation_id":"f82f58ef-f399-4ae2-8df7-c33c5e669b64","resolution":{"observed_at":"2026-08-09T20:50:17.577819Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.19258","last_updated":"2025-08-16T21:44:41Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-10T16:30:40.863297Z","submitted_at":"2025-01-31T16:16:52Z","title":"VisualSpeech: Enhancing Prosody Modeling in TTS Using Video"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":12,"verified_exact":0,"verified_fuzzy":23},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 1 inbound Pith citation observation for arXiv:2501.19258."}