{"as_of":"2026-08-10T12:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:51b0b4995f88ea5376a708701f8c3b9f0c86298fa5533f1b9476a623984b217c","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:21:11.541084Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-13T00:18:36.390365Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02742","snapshot_observed_at":"2026-07-13T00:18:36.390365Z","title":"Dake Guo, Xinfa Zhu, Liumeng Xue, Yongmao Zhang, Wenjie Tian, and Lei Xie","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.27376","last_updated":"2026-04-09T01:06:26Z","snapshot_observed_at":"2026-07-13T00:18:29.539736Z","submitted_at":"2026-04-09T01:06:26Z","title":"Unlocking Fine-Grained and Within-Utterance Speaking Style Control in Prompt-Based Text-to-Speech Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T00:18:36.390365Z"},"links":{"cited_paper":"/paper/2506.02742","citing_paper":"/paper/2605.27376"},"observation_digest":"sha256:20b0f34be19b7d97aecd4c8c01835cf954526edb21b1e21b7a1ce399c75b62e3","observation_id":"ab8d65e2-33a6-4319-b9f3-37bc6a94224c","resolution":{"observed_at":"2026-07-13T00:18:36.390365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02742/citation-record","integrity":"/paper/2506.02742/integrity","json":"/paper/2506.02742/citation-record.json","paper":"/paper/2506.02742"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.446538Z","title":"Text-to-speech synthesis based on latent variable conversion using diffusion probabilistic model and variational autoencoder,","venue":null,"work_id":"b4dc3907-bedb-43b2-a841-ade29e73cb89","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.227860Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:2a7d1bc9f013af927ecb9738d2748792a25be7709e4419ea06cce9b1b1c728b2","observation_id":"ecce2313-9455-4c17-a79e-d0cd65a7ac57","resolution":{"observed_at":"2026-08-07T11:21:14.449983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.435348Z","title":"A vector quantized approach for text to speech synthesis on real-world spontaneous speech,","venue":null,"work_id":"87cdafcf-c19e-419b-8fe4-05b023d6fd4c","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.283371Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b9f2e419f6e0596d5606865fce46613e186e49edba4b50d96af3cf08146385ef","observation_id":"711828ad-1b2d-48ca-bf52-46f70d9d2622","resolution":{"observed_at":"2026-08-07T11:21:14.438992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.423412Z","title":"Text to speech synthesis: A systematic review, deep learning based architecture and future research direction,","venue":null,"work_id":"da83eca1-54a4-447b-aacb-0f6645de05d8","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.394854Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:d84a3d3b9d0e02bef2bcd2bcfeeb8f8dc6c67d632e006bd9db37abef1171b397","observation_id":"3c09f19e-818a-44e6-a4df-7c27647eb4c4","resolution":{"observed_at":"2026-08-07T11:21:14.427231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.412759Z","title":"A style control technique for hmm-based expressive speech synthesis,","venue":null,"work_id":"83fb7ce8-f8cc-43ac-a18b-c9fbf969cb40","year":2007},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.499770Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:53078430e36db940d218ccee77460fe8e8455862392d9cb33318f1ad9e22cd83","observation_id":"9ebae707-3b16-4da4-a1b0-e46f971d4c15","resolution":{"observed_at":"2026-08-07T11:21:14.416305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05187","last_updated":"2023-12-08T17:18:42Z","snapshot_observed_at":"2026-08-06T14:36:57.042438Z","submitted_at":"2023-12-08T17:18:42Z","title":"Seamless: Multilingual Expressive and Streaming Speech Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05187","snapshot_observed_at":"2026-08-07T11:21:08.605193Z","title":"Seamless: Multilingual expressive and streaming speech translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.605193Z"},"links":{"cited_paper":"/paper/2312.05187","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:6df1ad95e7b452ebb1539d07aa11c60e30e165079bacf09be2559e4627140139","observation_id":"6161839f-97d8-4820-b17f-79b608a8c284","resolution":{"observed_at":"2026-08-07T11:21:08.605193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.401648Z","title":"Emotional speech synthesis with rich and granularized control,","venue":null,"work_id":"c9013927-b674-4a5f-85b4-03252cfbf9ae","year":2020},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.728143Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:1df064f04da332c49ed268664b44aea400367826027dfe3b866af56adcbc2a6d","observation_id":"256c2a04-a96e-4bb4-ae67-ecbeac21ffe0","resolution":{"observed_at":"2026-08-07T11:21:14.405471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.389841Z","title":"Emospeech: guiding fastspeech2 towards emotional text to speech,","venue":null,"work_id":"37c8f573-c8ce-450b-8360-d3953ca4a8de","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.786675Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:9fd550968acbc4389c3f6b9045a971460af43ae76183c44e552721a028213ee5","observation_id":"f993f611-a2c7-49cf-ac88-d101dd4d335b","resolution":{"observed_at":"2026-08-07T11:21:14.393376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05447","last_updated":"2017-11-28T02:07:43Z","snapshot_observed_at":"2026-07-06T06:09:32.998168Z","submitted_at":"2017-11-15T08:27:35Z","title":"Emotional End-to-End Neural Speech Synthesizer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05447","snapshot_observed_at":"2026-08-07T11:21:08.876739Z","title":"Emotional end-to-end neural speech synthesizer,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.876739Z"},"links":{"cited_paper":"/paper/1711.05447","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:78b65ee5cc627dcef7c24412e4fd02063c1d5a2dd3bba730c136d4dfeaacd1a9","observation_id":"6fe71321-f5c4-45f9-9aa5-b5428c8438d4","resolution":{"observed_at":"2026-08-07T11:21:08.876739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18398","last_updated":"2025-02-18T21:39:25Z","snapshot_observed_at":"2026-07-06T18:06:49.190172Z","submitted_at":"2024-04-29T03:19:39Z","title":"UMETTS: A Unified Framework for Emotional Text-to-Speech Synthesis with Multimodal Prompts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18398","snapshot_observed_at":"2026-08-07T11:21:08.945087Z","title":"Mm-tts: A unified framework for multimodal, prompt-induced emotional text-to- speech synthesis,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:08.945087Z"},"links":{"cited_paper":"/paper/2404.18398","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:27d379163f688bf0113e47f0bf1d9f925bd64e16b22ddfae41c36494c09a10a7","observation_id":"f07309b6-b2ef-48d8-b8e6-9e074049233b","resolution":{"observed_at":"2026-08-07T11:21:08.945087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.377719Z","title":"Emodiff: Intensity controllable emotional text-to-speech with soft-label guidance,","venue":null,"work_id":"19a6aa7f-b8d4-4bee-bc99-5e3cda16dd81","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.022798Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:fd15681ccc0c2abc9a88bd5eaf1cf7eac797e69ec99dbd486a55aa95d06070e5","observation_id":"1a1a03c4-ca9c-4191-b63d-0ab860352e53","resolution":{"observed_at":"2026-08-07T11:21:14.381827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:09.117066Z","title":"Instructtts: Modelling expressive tts in discrete latent space with natural language style prompt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.117066Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:010c130f9ee4e8585a814bb6f9a27cbfd9a81ae6a470800f84229e379d76a2b2","observation_id":"6cb213e9-216c-4e58-8c47-b84641a8b179","resolution":{"observed_at":"2026-08-07T11:21:09.117066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.358605Z","title":"Emotional voice conversion using dual supervised adversarial networks with continuous wavelet transform f0 features,","venue":null,"work_id":"fb3513d6-6d20-4d58-ae71-45c9094b52f4","year":2019},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.168520Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:7bcd11639480500a0f1d6895a2ea29f5e6f7e29af45482fb1cc0e6366e84d5e2","observation_id":"2893bc0f-cf81-40fb-a624-1adc0a30a85d","resolution":{"observed_at":"2026-08-07T11:21:14.361993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.347185Z","title":"Emotion controllable speech synthesis using emotion-unlabeled dataset with the assistance of cross-domain speech emotion recognition,","venue":null,"work_id":"4c68d45c-425d-4954-a061-9c904c5b7fab","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.260148Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:5b197e73a74be21e83803e6435707051ec0b91b24859da2b5ff02b717a3018e1","observation_id":"adf632af-02ca-46bf-9bcd-7d4ec723cb9b","resolution":{"observed_at":"2026-08-07T11:21:14.351042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12229","last_updated":"2024-09-17T10:40:11Z","snapshot_observed_at":"2026-07-06T18:47:33.863041Z","submitted_at":"2024-07-17T00:54:15Z","title":"Laugh Now Cry Later: Controlling Time-Varying Emotional States of Flow-Matching-Based Zero-Shot Text-to-Speech","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12229","snapshot_observed_at":"2026-08-07T11:21:09.341844Z","title":"Laugh now cry later: Controlling time-varying emotional states of flow-matching-based zero-shot text-to- speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.341844Z"},"links":{"cited_paper":"/paper/2407.12229","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:69548eafe685069abb8e23f773d4027e4f2691444d3292f6b5e680141fb815b5","observation_id":"60428489-0cc4-4ea2-a7a0-e3a977f91521","resolution":{"observed_at":"2026-08-07T11:21:09.341844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.00436","last_updated":"2022-10-06T07:47:16Z","snapshot_observed_at":"2026-08-08T00:25:41.908794Z","submitted_at":"2021-04-01T12:40:07Z","title":"Expressive Text-to-Speech using Style Tag","version":2},"cited_work":{"arxiv_id":"2104.00436","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.00436","snapshot_observed_at":"2026-08-07T11:21:12.472939Z","title":"Expressive Text-to-Speech using Style Tag","venue":"eess.AS","work_id":"60cef777-e48e-430c-bc98-a78d8e526aef","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.399942Z"},"links":{"cited_paper":"/paper/2104.00436","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:831d71f9ff4e35571492eb159e9f5a177bb614283362f26831311867363f9ea5","observation_id":"3c9a0693-f18e-4c09-9302-fe359de62c9e","resolution":{"observed_at":"2026-08-07T11:21:12.618602Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.335541Z","title":"Controllable emotion transfer for end-to-end speech synthesis,","venue":null,"work_id":"261fba5f-8203-472b-afc1-fea830a66362","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.472161Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:a3cd315d3c9d7a6bc5a96f9bc5b1c43627f159a71f67d1637d1cc2cda964e6be","observation_id":"b68b8e23-6532-4930-b0e2-83071a623862","resolution":{"observed_at":"2026-08-07T11:21:14.339093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15185","last_updated":"2023-12-23T07:46:55Z","snapshot_observed_at":"2026-08-08T00:25:44.100410Z","submitted_at":"2023-12-23T07:46:55Z","title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15185","snapshot_observed_at":"2026-08-07T11:21:09.566555Z","title":"emotion2vec: Self-supervised pre-training for speech emotion repre- sentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.566555Z"},"links":{"cited_paper":"/paper/2312.15185","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:15bbeaa828cf4695fabb5485a326c3298a32c5992feef212ce9e6a5750ee6de6","observation_id":"17b384e8-dc7f-4680-a281-53e03cb3b570","resolution":{"observed_at":"2026-08-07T11:21:09.566555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.323423Z","title":"Ttslow: Slow down text-to-speech with efficiency robustness evaluations,","venue":null,"work_id":"89d7f4f3-d2a2-4577-a8b9-4a7005fb3b64","year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.625831Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:fe40f1f1b362d39d5a9267b30af7b548fa900ab1abdae510ea358e5a6b044253","observation_id":"083825dc-034c-4b32-aad5-95ecadc7bf62","resolution":{"observed_at":"2026-08-07T11:21:14.327686Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12498","last_updated":"2025-06-22T16:51:47Z","snapshot_observed_at":"2026-08-08T00:26:32.968714Z","submitted_at":"2024-12-17T03:02:05Z","title":"Hierarchical Control of Emotion Rendering in Speech Synthesis","version":3},"cited_work":{"arxiv_id":"2412.12498","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.12498","snapshot_observed_at":"2026-08-07T11:21:12.229573Z","title":"Hierarchical Control of Emotion Rendering in Speech Synthesis","venue":"cs.SD","work_id":"7e93314b-8382-4d9f-8a27-cb4427e71650","year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.698405Z"},"links":{"cited_paper":"/paper/2412.12498","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:6833fc11e626db29c82bef33bb5d2bf1c750028e1ae1ee37308718fbdd4fadf3","observation_id":"4473eeaa-47f2-4a02-bada-4098ec1aec00","resolution":{"observed_at":"2026-08-07T11:21:12.319458Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.312310Z","title":"The nature of emotions: Human emotions have deep evolutionary roots, a fact that may explain their complexity and provide tools for clinical practice,","venue":null,"work_id":"4dde570c-c8ab-4451-9242-7cbeaedefd41","year":2001},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.755022Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:f482150ab45d23d6f625e996d0aaefb63e1be67651ae23e105e1daceb6a197be","observation_id":"74199a7f-3414-4434-aa5e-12c14f82761b","resolution":{"observed_at":"2026-08-07T11:21:14.316055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.301327Z","title":"Mixed emotions and coping: The benefits of secondary emotions,","venue":null,"work_id":"b342b21f-8cc0-46ac-9fac-9028179aebec","year":2014},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.827686Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:c8946f7006e8fb5c813a4de8970569db07f0828d95718ff1f8dfc9ab74b5febb","observation_id":"705b1bd0-4113-4a5a-ab4f-9d2b36ff4cf6","resolution":{"observed_at":"2026-08-07T11:21:14.304965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00648","last_updated":"2023-06-01T13:14:56Z","snapshot_observed_at":"2026-07-06T15:36:33.788201Z","submitted_at":"2023-06-01T13:14:56Z","title":"EmoMix: Emotion Mixing via Diffusion Models for Emotional Speech Synthesis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00648","snapshot_observed_at":"2026-08-07T11:21:09.901971Z","title":"Emomix: Emotion mixing via diffusion models for emotional speech synthesis,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:09.901971Z"},"links":{"cited_paper":"/paper/2306.00648","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:6b09ef1bfbc23898117b34a52b41dc19e253685be36a2231c63de45326929e01","observation_id":"4bc94bfc-c0c3-4c33-b7a0-fcc02be94234","resolution":{"observed_at":"2026-08-07T11:21:09.901971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:14.285295Z","title":"Speech synthe- sis with mixed emotions,","venue":null,"work_id":"4f8f8dca-d3d4-4cda-9598-eff27847471c","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.007123Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:305f7b2b92ddf2ebfa38ce10140ff62494965c12895acbeb1b3d6f273be40b9e","observation_id":"637c7ca9-a238-459a-b47b-d1b966943277","resolution":{"observed_at":"2026-08-07T11:21:14.293890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.980370Z","title":"Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis,","venue":null,"work_id":"b0ec0d2b-ff29-4154-92a3-60de6447e699","year":2018},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.082748Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b4cfa06c16337b97f89c4ec63ce33ded0f7f1e5d258f615498e3877c4bd0f927","observation_id":"e6bc07f3-ae9e-48a9-90e5-0a18379650d9","resolution":{"observed_at":"2026-08-07T11:21:14.148867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.911370Z","title":"Fine-grained emotion strength transfer, control and prediction for emotional speech synthesis,","venue":null,"work_id":"7a851078-fd67-4378-ae7d-d2aba5b93249","year":2021},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.142638Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:6f1ec29d640a33ed5ad9726367e0c64d9f1abd5e21e8fa1d90435bb3e6e21682","observation_id":"9c7fe274-f540-4bab-b00a-83df7318befd","resolution":{"observed_at":"2026-08-07T11:21:13.949555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.670011Z","title":"Emotion rendering for conversational speech synthesis with heterogeneous graph-based con- text modeling,","venue":null,"work_id":"4a8d7b74-446b-4d45-a3d0-4c7d002ff319","year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.225662Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:efe5d2c9fabb5f5e2cf5ba43fa882ccd5eb17a2c926ce3847bc0d731692f786a","observation_id":"2497f4b3-b964-42be-9c88-309eb5c181b8","resolution":{"observed_at":"2026-08-07T11:21:13.784289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.409644Z","title":"An emotion speech synthesis method based on vits,","venue":null,"work_id":"5f9a2a9b-aa00-4f65-963b-225053f63c15","year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.301961Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:5235cc43c7ad5557120ffc903eb11d2568188ebb071eff091465d16c54842eea","observation_id":"7d54fd94-5857-44f7-b432-56a2469d411b","resolution":{"observed_at":"2026-08-07T11:21:13.538624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:10.356461Z","title":"Emotional dimension control in language model-based text-to-speech: Spanning a broad spectrum of human emotions,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.356461Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:e13f56ce50a7c10000490bd2e68ed3c1e4e64e97972b8b435fd563fae604728a","observation_id":"755c1a2e-765d-4e1d-9ca4-84c51f3e0796","resolution":{"observed_at":"2026-08-07T11:21:10.356461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.10157","last_updated":"2024-09-16T10:41:36Z","snapshot_observed_at":"2026-08-09T10:31:00.505717Z","submitted_at":"2024-09-16T10:41:36Z","title":"Emo-DPO: Controllable Emotional Speech Synthesis through Direct Preference Optimization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.10157","snapshot_observed_at":"2026-08-07T11:21:10.425885Z","title":"Emo- dpo: Controllable emotional speech synthesis through direct preference optimization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.425885Z"},"links":{"cited_paper":"/paper/2409.10157","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:9b2d7c6e04d6d11dc57233c8013f71d56ef4a7d6bda1e0e51421eb9cbc9d146a","observation_id":"ea9f1639-a2a7-4b3d-9df5-8f402343c328","resolution":{"observed_at":"2026-08-07T11:21:10.425885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04128","last_updated":"2025-02-22T11:32:13Z","snapshot_observed_at":"2026-08-08T23:22:17.610458Z","submitted_at":"2025-02-06T15:04:00Z","title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04128","snapshot_observed_at":"2026-08-07T11:21:10.561460Z","title":"Llasa: Scaling train-time and inference- time compute for llama-based speech synthesis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.561460Z"},"links":{"cited_paper":"/paper/2502.04128","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:30abc2c307157d089d2143e91b86de5e81c0e9c82c7082f4af53f5052be1a68d","observation_id":"31081eed-7d6b-4f11-be56-b8fbb0f0f956","resolution":{"observed_at":"2026-08-07T11:21:10.561460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:13.197247Z","title":"Emo- dpo: Controllable emotional speech synthesis through direct preference optimization,","venue":null,"work_id":"5df9af8c-cd54-463e-b195-0e30e68bc4a4","year":2025},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.702413Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:2579822924ae9edc259c8b4eb664b371fd78266f7c63177dad52433832cec193","observation_id":"c0401735-66f1-420b-8b9c-bf5691d04bb4","resolution":{"observed_at":"2026-08-07T11:21:13.287159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:10.850442Z","title":"Plutchik and H","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.850442Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:fe5e82cd95d964efd6e0a6edd20faaad34c09214fd4b120877c2e240ca12c949","observation_id":"81984663-2d80-4764-902d-f31aca722a97","resolution":{"observed_at":"2026-08-07T11:21:10.850442Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:12.953074Z","title":"Cross and C","venue":null,"work_id":"2514adf7-b86d-4ff4-980d-97872902b527","year":2016},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.951853Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:3843a6364c03f0a53a5ae74997ea694af01ca1b9301e7ff71f353a66ab0057ee","observation_id":"bae3163c-c55d-4294-958b-5ca37cfbe1ec","resolution":{"observed_at":"2026-08-07T11:21:13.069996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.13756","last_updated":"2023-09-18T01:54:31Z","snapshot_observed_at":"2026-08-09T06:13:33.464070Z","submitted_at":"2022-10-25T03:50:34Z","title":"Mixed-EVC: Mixed Emotion Synthesis and Control in Voice Conversion","version":3},"cited_work":{"arxiv_id":"2210.13756","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.13756","snapshot_observed_at":"2026-08-07T11:21:11.776569Z","title":"Mixed-EVC: Mixed Emotion Synthesis and Control in Voice Conversion","venue":"eess.AS","work_id":"bfaecd31-5ac8-4fa0-9ccb-f0d9a4b24a41","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:10.966398Z"},"links":{"cited_paper":"/paper/2210.13756","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:1f676c418cd619942bc0b6f00389b6aba5e2020d7628171b39a82712d51954a3","observation_id":"c5aa09e8-0832-4615-8684-2e077fc93150","resolution":{"observed_at":"2026-08-07T11:21:11.919290Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T11:21:11.058975Z","title":"GPT-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.058975Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:b939aec151a56f92a955397008411fd3bc3c5186bf5066f87aceb9d3ef18655f","observation_id":"c99d1b9d-5204-4302-a65a-863817ae65f3","resolution":{"observed_at":"2026-08-07T11:21:11.058975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-07T11:21:11.145489Z","title":"Gemini: a family of highly capable multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.145489Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:dc2c04fe67fd97ad973057c1dff4104194f226996ca89ee042403e639ee1a32e","observation_id":"3ac0bed4-f7b3-4dea-96ab-d39ab32481a3","resolution":{"observed_at":"2026-08-07T11:21:11.145489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-07-06T18:42:34.958119Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T11:21:11.313415Z","title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.313415Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:761aa22432aed0a2080b8ae61e6f9935ef3deef00a9e674f771d83579235b546","observation_id":"3943616c-b04d-4732-b3eb-f60faec16bfc","resolution":{"observed_at":"2026-08-07T11:21:11.313415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:21:12.775881Z","title":"Emotional voice conversion: Theory, databases and esd,","venue":null,"work_id":"f5ac3f13-2bd6-403d-82a5-12cce5ec57dd","year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.425075Z"},"links":{"citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:28b1a9a4a4ef784b75f52568b9bd80e8a10a1fc2bf6c75373771096a867bb9d2","observation_id":"b63597ec-abaa-4a80-9499-ab4411a7ab91","resolution":{"observed_at":"2026-08-07T11:21:12.866746Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-07T11:21:11.541084Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:21:11.541084Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2506.02742"},"observation_digest":"sha256:a976781b7d8ece21f8f9789953eaae907e6431a15096ac846ea06dac597a1c0b","observation_id":"000021e7-1a98-488e-a767-fbe79ea35c0a","resolution":{"observed_at":"2026-08-07T11:21:11.541084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02742","last_updated":"2025-06-03T10:59:22Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-08T08:10:54.994072Z","submitted_at":"2025-06-03T10:59:22Z","title":"Prompt-Unseen-Emotion: Zero-shot Expressive Speech Synthesis with Prompt-LLM Contextual Knowledge for Mixed Emotions"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":3,"verified_fuzzy":21},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2506.02742."}