{"as_of":"2026-08-23T15:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3ca681c5fc76a85535c96b36fb5368173921d82bd0f1835ecf6e8a650fdea303","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T16:37:25.405095Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.13314/citation-record","integrity":"/paper/2411.13314/integrity","json":"/paper/2411.13314/citation-record.json","paper":"/paper/2411.13314"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"1703.10135","last_updated":"2017-04-06T21:20:34Z","snapshot_observed_at":"2026-08-17T12:01:30.659594Z","submitted_at":"2017-03-29T16:55:13Z","title":"Tacotron: Towards End-to-End Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1703.10135","snapshot_observed_at":"2026-08-12T16:37:25.246134Z","title":"Tacotron: Towards end-to- end speech synthesis,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.246134Z"},"links":{"cited_paper":"/paper/1703.10135","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:37a10c7fa058942f75155fc42579a701642e936ae962676ce3391f2489674f0e","observation_id":"dc4ebf00-4150-4aec-87fe-542f67aadc66","resolution":{"observed_at":"2026-08-12T16:37:25.246134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.942424Z","title":"Fastspeech: Fast, robust and controllable text to speech,","venue":null,"work_id":"42661964-bd8e-4f55-a0e9-5440362e87a4","year":2019},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.252316Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:e4a038bb585884486b4a403e1ca67398efd4f3a1600ffe47d98cd2e573199c69","observation_id":"eb9b04be-2296-416d-b495-5f8221359ae1","resolution":{"observed_at":"2026-08-12T16:37:25.947607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.927705Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"93fb83c9-60d4-451e-a420-061d7ba1a23c","year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.257311Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:a7881ede1375057a5e165fb2d39984f1c2186b6e849c2dd8331c762fde154f2d","observation_id":"7c8fb9a9-9428-4907-8a89-668d732f99d2","resolution":{"observed_at":"2026-08-12T16:37:25.932600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-12T16:37:25.262204Z","title":"Neural codec language models are zero-shot text to speech synthesizers,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.262204Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:96dde8dabf4741896953eb9bab47d3743ba18d3c7dc6cb3babc266fa59a997a7","observation_id":"fe39d2f6-fcc0-44bd-954b-b91685061290","resolution":{"observed_at":"2026-08-12T16:37:25.262204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.912359Z","title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":"4425b3b9-d305-4ff2-94c6-7f3181c66a1e","year":2022},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.268301Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:90ca2a8fc53820df620a29076313db12c2ac9d3f57012630030f37922b0c10ee","observation_id":"dee475ef-774e-4f1e-b786-40bf56f2d4f4","resolution":{"observed_at":"2026-08-12T16:37:25.917378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.897441Z","title":"Prompttts: Controllable text-to-speech with text descriptions,","venue":null,"work_id":"c876175d-7fb8-4c0d-a1a1-ca605353648d","year":2023},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.273408Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:84201f951128cec2098f54ac6dc6818073257a01a2637303d7121bd6e3beec04","observation_id":"169871b4-4acf-47c2-a6b8-b227e7d0150f","resolution":{"observed_at":"2026-08-12T16:37:25.902242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.882664Z","title":"Instructtts: Mod- elling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":"db2f278c-054f-4b2c-8301-cdb1f9c4e448","year":2024},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.278929Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:e7703f9bdee767579f699d98c0216af25175481c4394a5915b8ed16488383186","observation_id":"9a51dbef-f8e1-434d-b4d1-17e8bc538ca2","resolution":{"observed_at":"2026-08-12T16:37:25.887551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.867677Z","title":"Mm-tts: Multi-modal prompt based style transfer for expres- sive text-to-speech synthesis,","venue":null,"work_id":"74ec7e46-814d-407d-966e-f671a090e672","year":2024},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.283736Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:7cd1d72db2271ea217eb2ecaffc479b7daf21565abd5d976087743a2ebd2e500","observation_id":"aca669cf-095d-49cd-95a4-46bece36ba6c","resolution":{"observed_at":"2026-08-12T16:37:25.872560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.852407Z","title":"Comparison of different impulse response measurement techniques,","venue":null,"work_id":"743f7073-b1d5-4714-9be0-3413f432d2c2","year":2002},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.288410Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:c20623eb2ade55a9c883ca1fa953d191ce42e62dffe9ed43bbc465d2a5ab490e","observation_id":"75601def-1764-431c-8831-9d72931c4282","resolution":{"observed_at":"2026-08-12T16:37:25.857455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.03887","last_updated":"2022-08-06T18:55:53Z","snapshot_observed_at":"2026-08-17T00:49:55.627599Z","submitted_at":"2021-10-08T04:19:19Z","title":"Environment Aware Text-to-Speech Synthesis","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.03887","snapshot_observed_at":"2026-08-12T16:37:25.293133Z","title":"Environment aware text-to-speech synthesis,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.293133Z"},"links":{"cited_paper":"/paper/2110.03887","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:8832a24b002b09dd6c7074ef09b20a8c59090fb36d5b463653cd8741e08c14bf","observation_id":"1bb2be7d-8c94-4984-b95e-3187ed4fd516","resolution":{"observed_at":"2026-08-12T16:37:25.293133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.837416Z","title":"V oiceldm: Text-to-speech with environmental context,","venue":null,"work_id":"d099d8d3-93dc-421e-b43d-726e4141b85b","year":2024},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.298076Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:9f9893aa7af63401c94cbec8061d761a9056ee25cca6f1c505f5fc5579c2ceaa","observation_id":"637f170c-7109-4377-82ee-998ae92e25df","resolution":{"observed_at":"2026-08-12T16:37:25.842304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12708","last_updated":"2024-04-21T07:17:14Z","snapshot_observed_at":"2026-08-21T15:04:44.573642Z","submitted_at":"2023-05-22T04:37:41Z","title":"ViT-TTS: Visual Text-to-Speech with Scalable Diffusion Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12708","snapshot_observed_at":"2026-08-12T16:37:25.302545Z","title":"Vit-tts: visual text-to-speech with scalable diffusion transformer,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.302545Z"},"links":{"cited_paper":"/paper/2305.12708","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:6efcbed6fdac33c3bd6369e68954d0c2c7fb305ffd42b921d3a181c78d75beb4","observation_id":"05883d7e-61ef-4aca-80db-5b1cf7666636","resolution":{"observed_at":"2026-08-12T16:37:25.302545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.822319Z","title":"Multi-source spatial knowledge understanding for immersive visual text-to-speech,","venue":null,"work_id":"31e20eff-3cfa-4871-8371-571e06598f2a","year":2025},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.307222Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:845f9416ac045b9f7ce1f0824e28207d9efc83f74026010388bc0a101c04c0f4","observation_id":"75755e0f-adfb-430e-b53f-29d599cada24","resolution":{"observed_at":"2026-08-12T16:37:25.827568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.807592Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"31cf9b50-e0e0-46e6-a0f9-b42283a81807","year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.311684Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:95bcda1751fe4f132aa0fd2dcfefce0ac4ccc8f35f856db4778b60259692caaa","observation_id":"5830bfac-aef0-4a35-acc2-faef4795b83a","resolution":{"observed_at":"2026-08-12T16:37:25.812344Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.792573Z","title":"Allen, M","venue":null,"work_id":"34810091-aee4-45ae-8204-2ebf6aacaf37","year":1987},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.316049Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:6b7df535c3b6393679a26181ed41813460e9be9c37f1cdaaec4005896ffcad27","observation_id":"26eab282-68aa-40dd-bb41-bbb545d206aa","resolution":{"observed_at":"2026-08-12T16:37:25.797123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.777632Z","title":"The festival speech synthesis system,","venue":null,"work_id":"5485d05f-9a43-472a-90a3-83bde54c79eb","year":1998},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.320515Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:f162ad5c5bb1e9f9a08418eb971d354fc902b9e01b1c9d89fe2a065950da135a","observation_id":"40f962a1-9329-4381-ae5a-14b7e13fb4bc","resolution":{"observed_at":"2026-08-12T16:37:25.782323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.762204Z","title":"An hmm-based speech synthesis system applied to english,","venue":null,"work_id":"353dba5f-d0f6-479c-9afb-ac356817aa2d","year":2002},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.325063Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:9b17fce12aab90bc5568a58eeb528ec54b84897418cbf66d69a6cc58a1226b07","observation_id":"4296abba-49b2-4dc7-9102-fc2e7cfde937","resolution":{"observed_at":"2026-08-12T16:37:25.767141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04558","last_updated":"2022-08-08T01:53:05Z","snapshot_observed_at":"2026-08-20T03:31:53.345681Z","submitted_at":"2020-06-08T13:05:40Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04558","snapshot_observed_at":"2026-08-12T16:37:25.329513Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.329513Z"},"links":{"cited_paper":"/paper/2006.04558","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:7b24d1d16664264a0482a039c404353cedb157426ed45d5d1fd25a36e70e27d5","observation_id":"7d176ddb-9b26-4ff6-935c-4a9d5e2baec5","resolution":{"observed_at":"2026-08-12T16:37:25.329513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01409","last_updated":"2021-04-03T13:53:19Z","snapshot_observed_at":"2026-08-23T14:00:23.790748Z","submitted_at":"2021-04-03T13:53:19Z","title":"Diff-TTS: A Denoising Diffusion Model for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.01409","snapshot_observed_at":"2026-08-12T16:37:25.334264Z","title":"Diff- tts: A denoising diffusion model for text-to-speech,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.334264Z"},"links":{"cited_paper":"/paper/2104.01409","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:40c83b3e90ebe3908a95691e4668793df3071b32dd84c57181219577cdd968f6","observation_id":"69b0d3b4-99c5-4976-9e38-e90db2844ca6","resolution":{"observed_at":"2026-08-12T16:37:25.334264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.745917Z","title":"Prodiff: Progressive fast diffusion model for high-quality text-to-speech,","venue":null,"work_id":"0cdd3f35-11ac-4bde-83a6-66223ea8c1d5","year":2022},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.339265Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:50684f045a886335f71b5ab91ae9dceae74bd5e81d9a584ec20b4c315633b183","observation_id":"c29923ea-f3d5-4bc9-9ff8-efa8f60177a6","resolution":{"observed_at":"2026-08-12T16:37:25.751216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.728879Z","title":"Clam-tts: Improving neural codec language model for zero-shot text-to-speech,","venue":null,"work_id":"1b6b26a4-c586-4d60-b096-6d72f3e547c0","year":2023},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.343956Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:df79c2462bba3add88fcea32086e344f37fe36571f1bb9cd7fd8aeeea1c4b613","observation_id":"04fff5d5-02ef-4d4a-8020-d429f7929f39","resolution":{"observed_at":"2026-08-12T16:37:25.733999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.713209Z","title":"Audiolm: a language modeling approach to audio generation,","venue":null,"work_id":"0457d68f-fdef-4e6a-86b9-6da7adcbf502","year":2023},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.348479Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:75d5a945fee3b6ecc5ae72919c2dea962257a2ac1bbcc97469bad0c503a5503f","observation_id":"b67974d1-4296-4cfe-b0c3-1bff148f1c4e","resolution":{"observed_at":"2026-08-12T16:37:25.718300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03100","last_updated":"2024-04-23T08:38:03Z","snapshot_observed_at":"2026-08-18T17:50:58.009731Z","submitted_at":"2024-03-05T16:35:25Z","title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03100","snapshot_observed_at":"2026-08-12T16:37:25.353064Z","title":"Naturalspeech 3: Zero-shot speech syn- thesis with factorized codec and diffusion models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.353064Z"},"links":{"cited_paper":"/paper/2403.03100","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:7d789a1346b3830bfee5e528a4a48ab2dad05177aedf35286670a062b20326db","observation_id":"8685cbbd-04a9-4cf4-8f11-372eed8bc125","resolution":{"observed_at":"2026-08-12T16:37:25.353064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.696603Z","title":"Image2reverb: Cross-modal reverb impulse response synthesis,","venue":null,"work_id":"bdc79366-d632-4894-8b6f-9d9ccbd8bfb2","year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.357945Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:e755475630d6de14fcb43745545546234951d01239d0a40fade57082a08da5fa","observation_id":"7d6ad148-2fe3-4a92-b82b-b538d5dbd9c2","resolution":{"observed_at":"2026-08-12T16:37:25.702539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.680978Z","title":"Visual acoustic match- ing,","venue":null,"work_id":"38f68e2f-a7a1-4f99-b622-1942e71ba37c","year":2022},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.362533Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:c38d295120de816adabff4c1704b428188f8c8b1b19e5651e1435ce56571e4b3","observation_id":"569aa7d9-a979-4b90-bf14-a7dfb4fcf717","resolution":{"observed_at":"2026-08-12T16:37:25.685836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.665040Z","title":"Self-supervised visual acoustic matching,","venue":null,"work_id":"1bf105fe-3cc7-49e0-a8ab-42fe88df02d0","year":2024},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.367038Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:c44f54891adc6ba3ce92bb5f97479f0bb4a089a039ba7559e5ff73f7a005b9a6","observation_id":"e95df2cb-c3c2-44db-b18b-71e25a8d0d24","resolution":{"observed_at":"2026-08-12T16:37:25.670026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.371754Z","title":"Meta-stylespeech: Multi- speaker adaptive text-to-speech generation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.371754Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:aae6ab338d997608000514cb6b05c6fd12788257d86df3980e7b43ef62be5e2b","observation_id":"e4a73fb9-314e-453b-a55e-b1128a46360d","resolution":{"observed_at":"2026-08-12T16:37:25.371754Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.376435Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.376435Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:94d1a95bb6d0c1569b39454499b63846a90d5f9894ebc2a4214127231c8e7a5a","observation_id":"57329459-da0e-4a55-9d58-18eb0eea6e71","resolution":{"observed_at":"2026-08-12T16:37:25.376435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.381217Z","title":"The lj speech dataset,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.381217Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:256d51b75de5766f733ec6195e5d7a1c4793a42d8f7711eb168ad9b5e0bf1853","observation_id":"71fe08eb-bd37-4259-a8b9-16a3fc517408","resolution":{"observed_at":"2026-08-12T16:37:25.381217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.618604Z","title":"Cstr vctk corpus: English multi-speaker corpus for cstr voice cloning toolkit,","venue":null,"work_id":"f2b21537-2164-466d-9c4b-dd64c1ad2655","year":2017},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.386009Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:d6a74ffca059cd97b90ce9db94f88d5f48b2670e7ce807a3d9490d2a201434d4","observation_id":"96254a7e-cff5-49b0-b8f6-af0bdff4fb4c","resolution":{"observed_at":"2026-08-12T16:37:25.623776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.602604Z","title":"Mel-cepstral distance measure for objective speech quality assessment,","venue":null,"work_id":"ef5b42a1-b27a-45ed-ae4a-5098697f0cf4","year":1993},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.390762Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:db84db046812ac7b97e20855ab7aec8b06ba25f099cd096515008d7f56ba0652","observation_id":"43e85903-b0a8-459b-b8c9-f919c15bc58b","resolution":{"observed_at":"2026-08-12T16:37:25.607783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.05557","last_updated":"2021-06-15T22:19:50Z","snapshot_observed_at":"2026-08-16T18:34:08.571873Z","submitted_at":"2021-04-02T22:31:45Z","title":"SC-GlowTTS: an Efficient Zero-Shot Multi-Speaker Text-To-Speech Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.05557","snapshot_observed_at":"2026-08-12T16:37:25.395526Z","title":"Sc-glowtts: An efficient zero-shot multi-speaker text-to-speech model,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.395526Z"},"links":{"cited_paper":"/paper/2104.05557","citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:4bd46d9b1b766e0a1739804004685917e79733bb86eba2d143be80f5d66064a8","observation_id":"46dacfa0-6931-4e79-81dd-31c6803a785f","resolution":{"observed_at":"2026-08-12T16:37:25.395526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.585856Z","title":"An overview of voice con- version and its challenges: From statistical modeling to deep learning,","venue":null,"work_id":"72c0d0a5-d681-416b-9c55-f170e31a14ec","year":2020},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.400582Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:41add13e8667482ebd35fabdead2f91a7afeaf9e679cdc5d5b1b66420af6e873","observation_id":"4883fca5-8629-48f6-ade2-2c9f514ef95d","resolution":{"observed_at":"2026-08-12T16:37:25.591490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T16:37:25.566903Z","title":"Grad-cam: Visual explanations from deep networks via gradient-based localization,","venue":null,"work_id":"7c440a5a-90d6-43e5-ae82-c4aa118464b6","year":2017},"citing_paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception","version":4},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T16:37:25.405095Z"},"links":{"citing_paper":"/paper/2411.13314"},"observation_digest":"sha256:dfc0ca94b531ab535eb069d8c0f9dd35800a75b51c80ace9e82a07cacd6641a2","observation_id":"da33eb71-cba3-4e7f-a4dd-3596581e09f8","resolution":{"observed_at":"2026-08-12T16:37:25.574200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.13314","last_updated":"2025-09-03T04:42:41Z","latest_version":4,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-20T07:39:50.049191Z","submitted_at":"2024-11-20T13:28:42Z","title":"I2TTS: Image-indicated Immersive Text-to-speech Synthesis with Spatial Perception"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":23},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 0 inbound Pith citation observations for arXiv:2411.13314."}