{"as_of":"2026-08-10T04:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:31e7b3cb13b746eb0b17fe7b0cf77ed40ccf07481d87f30247238b7ac8257bdf","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:43:52.529754Z","state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.20945/citation-record","integrity":"/paper/2506.20945/integrity","json":"/paper/2506.20945/citation-record.json","paper":"/paper/2506.20945"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.842383Z","title":"Mega- tts 2: Boosting prompting mechanisms for zero-shot speech synthesis,","venue":null,"work_id":"6f69a68c-90a3-417e-aef2-e03f88367c37","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.228144Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:b9bc0448942c419305d04834b99eeafe4e669d786497be7822d2a6bb416b9323","observation_id":"27adb6b3-b24d-4ae4-8078-4cbcf618dc2c","resolution":{"observed_at":"2026-08-06T22:43:55.933915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-06T22:43:50.265834Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.265834Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:0eaa71e4c1e169c8cfdab3cbe55f5f4f75b21071ee83c301462a7c00e77e3364","observation_id":"49a3a70a-a4b9-4624-8064-993c0af597aa","resolution":{"observed_at":"2026-08-06T22:43:50.265834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00750","last_updated":"2024-10-20T14:25:49Z","snapshot_observed_at":"2026-08-01T10:20:49.367508Z","submitted_at":"2024-09-01T15:26:30Z","title":"MaskGCT: Zero-Shot Text-to-Speech with Masked Generative Codec Transformer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00750","snapshot_observed_at":"2026-08-06T22:43:50.332197Z","title":"Maskgct: Zero-shot text-to-speech with masked generative codec transformer,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.332197Z"},"links":{"cited_paper":"/paper/2409.00750","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:eef7186a3013a40039341b10dc054ddcf86896be8b916bad8dab164c99da80db","observation_id":"663c1c04-c3e4-47e7-9303-e95d307ed5f0","resolution":{"observed_at":"2026-08-06T22:43:50.332197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.634364Z","title":"Imaginary voice: Face-styled diffusion model for text-to-speech,","venue":null,"work_id":"051a348d-6c31-4f0c-9be7-9afe024c670b","year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.379308Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:7d7f174869bd56db76cff87e34bf3c0a33955c9ff90d25e729d9e01b1feb8fb4","observation_id":"80bebff5-0883-408a-bb4e-bb46906172b1","resolution":{"observed_at":"2026-08-06T22:43:55.768449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.478421Z","title":"SYNTHE-SEES: Face based text-to-speech for virtual speaker,","venue":null,"work_id":"3573fd3d-bf0c-4564-9dac-d417fc5cb024","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.454168Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:df57f3d7b0b8e176db0473af8352f0b52837c3bce476909de1bb5ddbff79c4d4","observation_id":"3bae3fb5-4477-40bf-a466-d1206948ef89","resolution":{"observed_at":"2026-08-06T22:43:55.560254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.313699Z","title":"Face2Speech: Towards multi-speaker text-to-speech synthesis using an embedding vector predicted from a face image.,","venue":null,"work_id":"429a0ed0-7c91-4f90-a9c4-7e7ce77fdeeb","year":2020},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.528613Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:a21c4000af8d2cce50ab3c2fbe70543a4cad0381ae21588d2b84763fe6b6383b","observation_id":"921cc365-a7bf-4ae9-aafd-b477a96c2b58","resolution":{"observed_at":"2026-08-06T22:43:55.395195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.157886Z","title":"FVTTS : Face based voice synthesis for text-to-speech,","venue":null,"work_id":"f5785df7-3794-482c-88ce-fa86b250b42b","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.580539Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:8550fa8e763dc9709cb4478865b5f48db46ad38f33924b1ffcdab2714949d443","observation_id":"3cf14a4c-5247-4d3f-a3fd-edced867f33d","resolution":{"observed_at":"2026-08-06T22:43:55.239168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:55.027835Z","title":"Instructtts: Modelling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":"87ed606f-22cf-4b63-9b54-7c32c544093f","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.633123Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:568c3ddc8c529ab381a903a1505049a79777b4bd8dc75364d4602b7792f17f41","observation_id":"a53c088a-a93b-47ee-9a23-c88649a88f76","resolution":{"observed_at":"2026-08-06T22:43:55.088342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.860818Z","title":"Prompttts: Controllable text-to-speech with text descriptions,","venue":null,"work_id":"aa50650d-d6a9-487b-89d0-3b93e313927f","year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.701372Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:81e45018094b30d948d39fbdaee8cd5fc00e604d5f69dfa07b065df66ad8d0f1","observation_id":"ffd35756-5c7b-4282-9d92-0eb646be2e2d","resolution":{"observed_at":"2026-08-06T22:43:54.942659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.713808Z","title":"PromptTTS++: Controlling speaker identity in prompt-based text-to- speech using natural language descriptions,","venue":null,"work_id":"f59b1b09-9bf4-4ab3-932d-b80a7f1e03e5","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.766499Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:856c263971f434e836602eb8825fba2d948674f4069f893fdb35da4856aae4b2","observation_id":"3556fb19-c8c4-4690-a30d-2355895035fb","resolution":{"observed_at":"2026-08-06T22:43:54.781073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18398","last_updated":"2025-02-18T21:39:25Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T03:19:39Z","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18398","snapshot_observed_at":"2026-08-06T22:43:50.830879Z","title":"Mm-tts: A unified framework for multimodal, prompt-induced emotional text-to-speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.830879Z"},"links":{"cited_paper":"/paper/2404.18398","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:a7816575da7176c1cfb052d1c8f624c86aa606adeb1e0b94140792cd81c9d947","observation_id":"96b659af-c3d2-464f-83f5-8ed7af6292f7","resolution":{"observed_at":"2026-08-06T22:43:50.830879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.559656Z","title":"MM-TTS: Multi- modal prompt based style transfer for expressive text-to-speech synthe- sis,","venue":null,"work_id":"3dc12fc1-1c2e-43ce-82ad-828bbe1ef589","year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.902999Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:30abe443de3cebb61bc3ce799ebbd9c84ca87efe0d24f80f916725e85c984f5c","observation_id":"6de8182b-399d-45b8-bde7-dc17676d2a97","resolution":{"observed_at":"2026-08-06T22:43:54.631249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.424176Z","title":"Gen- eralized end-to-end loss for speaker verification,","venue":null,"work_id":"63193231-fdc2-4677-876f-133b6badaceb","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:50.984883Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:78d8eedbb8e94938379e6ce475e00aa41cb944883fad263a687c10ffb3c67f05","observation_id":"9b3f1fe2-a79f-49e2-ac7e-39310830e03f","resolution":{"observed_at":"2026-08-06T22:43:54.479581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.245956Z","title":"Additive margin softmax for face verification,","venue":null,"work_id":"33d1cf62-afcd-49da-b7c9-c81c2b9102bb","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.073749Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:0951add9b9ae5149e59f37b7045eb2d7ea13388f87fc93c0cf322441016492a7","observation_id":"42cd64a7-6d95-4511-bf4f-adf0c4a098f7","resolution":{"observed_at":"2026-08-06T22:43:54.350835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:54.063896Z","title":"Bridging the gap between object and image-level representations for open-vocabulary detection,","venue":null,"work_id":"2a3efe4c-cfb6-4972-bb40-912343438c3a","year":2022},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.143577Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:957a5114186e0eec6f33f48fe012905b48197eafb364e81e883bf7c7e8f1d828","observation_id":"513fb0e2-5d0e-4c9d-980a-02ad682a3cba","resolution":{"observed_at":"2026-08-06T22:43:54.148216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.897398Z","title":"Joint- teaching: Learning to refine knowledge for resource-constrained un- supervised cross-modal retrieval,","venue":null,"work_id":"7b2b6c54-b680-42e0-9d4c-71ebbd6e0da1","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.255727Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:697814013a2f8eb6196c32299215f208d387e7ca25abd74c43813b3d36604cf4","observation_id":"1a202008-871a-415c-9415-9faaffdcd3d8","resolution":{"observed_at":"2026-08-06T22:43:54.002813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-06T22:43:51.359236Z","title":"Represen- tation learning with contrastive predictive coding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.359236Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:f26433b3829dab317bc308d624ca1b3587951760a791a6908f812b775df09131","observation_id":"8de811c5-5e1f-49dc-865e-18f221ead62d","resolution":{"observed_at":"2026-08-06T22:43:51.359236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.713416Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"73bcf64b-1a33-4fbe-a33c-ab55d88c4b83","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.427727Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:9cafaf20b1056b3e02c3f88dca0a39291b4c88ba8533cc239c9b970a6d23dbce","observation_id":"283108dd-26b0-4501-ae07-6e229d58e159","resolution":{"observed_at":"2026-08-06T22:43:53.818807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1809.00496","last_updated":"2018-10-28T14:29:46Z","snapshot_observed_at":"2026-07-06T06:58:51.877372Z","submitted_at":"2018-09-03T08:38:34Z","title":"LRS3-TED: a large-scale dataset for visual speech recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1809.00496","snapshot_observed_at":"2026-08-06T22:43:51.542194Z","title":"Lrs3- ted: a large-scale dataset for visual speech recognition,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.542194Z"},"links":{"cited_paper":"/paper/1809.00496","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:c7f270db52bc63b7e183ab58261163e5cd59abf9e0ee475f7992556b5026e31e","observation_id":"a29d89ed-059f-4cef-a988-bf5424033794","resolution":{"observed_at":"2026-08-06T22:43:51.542194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.513270Z","title":"Multi- caption text-to-face synthesis: Dataset and algorithm,","venue":null,"work_id":"f2c8e99f-7766-41bd-a1aa-5eb1f26e6fca","year":2021},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.614760Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:bf9754734c7ff149c054e895ba84eb74ff3a8be7d005ed404bbf2c77985a7c07","observation_id":"4f76282e-dc9d-4e21-a18a-023dd6db6970","resolution":{"observed_at":"2026-08-06T22:43:53.588907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07969","last_updated":"2024-06-12T07:49:21Z","snapshot_observed_at":"2026-07-06T18:29:25.765578Z","submitted_at":"2024-06-12T07:49:21Z","title":"LibriTTS-P: A Corpus with Speaking Style and Speaker Identity Prompts for Text-to-Speech and Style Captioning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07969","snapshot_observed_at":"2026-08-06T22:43:51.707283Z","title":"LibriTTS-p: A corpus with speaking style and speaker identity prompts for text-to-speech and style caption- ing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.707283Z"},"links":{"cited_paper":"/paper/2406.07969","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:86ef9989070daebb7bf7fddb4ed0a210e257b481fa440771fb07919187d6c701","observation_id":"5eb4f155-d80c-43eb-b966-9739821a71fa","resolution":{"observed_at":"2026-08-06T22:43:51.707283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18802","last_updated":"2023-05-30T07:30:21Z","snapshot_observed_at":"2026-08-10T04:18:28.247428Z","submitted_at":"2023-05-30T07:30:21Z","title":"LibriTTS-R: A Restored Multi-Speaker Text-to-Speech Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18802","snapshot_observed_at":"2026-08-06T22:43:51.837350Z","title":"Libritts-r: A restored multi-speaker text-to-speech corpus,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.837350Z"},"links":{"cited_paper":"/paper/2305.18802","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:88785ebff2097d0a29acddfb5630b32ecfa2fdc75d73606d4223b0ddd4b8132a","observation_id":"d4b436d8-9b4e-4335-b459-f4b37136a926","resolution":{"observed_at":"2026-08-06T22:43:51.837350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.02882","last_updated":"2019-04-05T06:05:00Z","snapshot_observed_at":"2026-08-07T13:32:24.356336Z","submitted_at":"2019-04-05T06:05:00Z","title":"LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.02882","snapshot_observed_at":"2026-08-06T22:43:51.942109Z","title":"Libritts: A corpus derived from librispeech for text-to-speech,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:51.942109Z"},"links":{"cited_paper":"/paper/1904.02882","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:b1990dd57e3cd0b3f04186fdd5d997bd9ee8e03d7e230e7d18e0bb0c20141ccf","observation_id":"1eb94adb-cfed-44c4-8ab4-2efea73e8fff","resolution":{"observed_at":"2026-08-06T22:43:51.942109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.317074Z","title":"Facenet: A unified embedding for face recognition and clustering,","venue":null,"work_id":"59b03eea-707e-4a7a-9fb4-8f1f4f51f197","year":2015},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.028232Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:8dee3e3e30baf428cfa667c11dd3e991580773b4f13129729d157aa622b7bac4","observation_id":"30bb92a6-34a9-4db0-8670-fc1c27a258f0","resolution":{"observed_at":"2026-08-06T22:43:53.414214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:53.131288Z","title":"Vggface2: A dataset for recognising faces across pose and age,","venue":null,"work_id":"493da65b-b2d1-4f88-8147-9e45f9e6475f","year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.103593Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:60e470ab46799db9426d45d2eccff5030beeb3118bb0abd2eb88c0caf78f7861","observation_id":"a1b04ad8-8725-496d-a7bc-e9b50f752036","resolution":{"observed_at":"2026-08-06T22:43:53.221017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:52.934755Z","title":"Joint face detection and alignment using multitask cascaded convolutional networks,","venue":null,"work_id":"bfc79064-ab5a-4e55-867d-d17fa82dfadb","year":2016},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.210637Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:cc535e64620d40039707dbd6a5856b393e1734d3fc0851ce2313be6769fe0db1","observation_id":"0398b027-d2a8-43c4-b82e-21d5e5777b9b","resolution":{"observed_at":"2026-08-06T22:43:53.040774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-08-07T15:21:18.262728Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-08-06T22:43:52.286082Z","title":"Ecapa- tdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.286082Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:77d17ab758442065338fedbb4e6a25410acfeea82ee5efd2d45990d5c0edd490","observation_id":"0859996d-b0fe-41e4-8386-aaebfc5501b1","resolution":{"observed_at":"2026-08-06T22:43:52.286082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.05622","last_updated":"2018-06-27T01:49:17Z","snapshot_observed_at":"2026-08-08T04:36:13.309903Z","submitted_at":"2018-06-14T15:59:12Z","title":"VoxCeleb2: Deep Speaker Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.05622","snapshot_observed_at":"2026-08-06T22:43:52.418324Z","title":"V oxceleb2: Deep speaker recognition,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.418324Z"},"links":{"cited_paper":"/paper/1806.05622","citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:23346250e690a1c974c6e13b38523ebc513d0da042ce488ce0b0d1eb0ebfe813","observation_id":"5e0dcf08-eb5f-4c7c-b3d9-6e454ef07ed8","resolution":{"observed_at":"2026-08-06T22:43:52.418324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T22:43:52.754031Z","title":"Explor- ing the limits of transfer learning with a unified text-to-text transformer,","venue":null,"work_id":"f1eab33e-336f-40b9-9585-f1e67f453c67","year":2020},"citing_paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T22:43:52.529754Z"},"links":{"citing_paper":"/paper/2506.20945"},"observation_digest":"sha256:f271e17928b0be1ef07f775064827ce91b5f9f11ab05dc79305558820f28671a","observation_id":"385dc97e-106a-4cb1-b335-b4c575076666","resolution":{"observed_at":"2026-08-06T22:43:52.825995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.20945","last_updated":"2025-06-26T02:23:42Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-08T15:43:11.330421Z","submitted_at":"2025-06-26T02:23:42Z","title":"A Multi-Stage Framework for Multimodal Controllable Speech Synthesis"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":19},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 0 inbound Pith citation observations for arXiv:2506.20945."}