{"as_of":"2026-08-20T20:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e57d0afa617820c79eb26e9873a889ee90d569ab955b35c08e398a1d01b25b8d","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:00:35.341553Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:00:32.452371Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T17:00:35.528453Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.12015","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12015","snapshot_observed_at":"2026-08-06T17:00:35.528453Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","venue":"cs.SD","work_id":"1d007a0f-de58-4b72-a356-2d460a480fd4","year":2025},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.452371Z"},"links":{"cited_paper":"/paper/2507.12015","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:e183a02caabccb86f0345dfeb6d782fda3e1995696d269598f5b60e69a426eb5","observation_id":"a5be3642-8335-4aa2-bbcd-e0705651b4ea","resolution":{"observed_at":"2026-08-06T17:00:35.612183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.12015/citation-record","integrity":"/paper/2507.12015/integrity","json":"/paper/2507.12015/citation-record.json","paper":"/paper/2507.12015"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"cited_work":{"arxiv_id":"2507.12015","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12015","snapshot_observed_at":"2026-08-06T17:00:35.528453Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","venue":"cs.SD","work_id":"1d007a0f-de58-4b72-a356-2d460a480fd4","year":2025},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.452371Z"},"links":{"cited_paper":"/paper/2507.12015","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:e183a02caabccb86f0345dfeb6d782fda3e1995696d269598f5b60e69a426eb5","observation_id":"a5be3642-8335-4aa2-bbcd-e0705651b4ea","resolution":{"observed_at":"2026-08-06T17:00:35.612183Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.254619Z","title":"Overview The overall architecture of EME-TTS is shown in Figure 1a","venue":null,"work_id":"7cbedcf2-95f9-4f7e-b73f-ec0f462e7c9e","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.546382Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:19b59f9b8fac54c29b8634d086d830a1c744b91d3fa83c30ad15652f6e600506","observation_id":"f3fd56f6-044e-4b73-883b-f093917cbfa0","resolution":{"observed_at":"2026-08-06T17:00:41.294741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.019092Z","title":null,"venue":null,"work_id":"ac4025bc-74dc-409d-991f-d937face0b1a","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.720040Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ef610597ebbbc2386fbb0e62f0aec1ce5eb0694bd1e5793c848228662e2b6b20","observation_id":"ac11db75-a129-483a-a4e5-ffa6d14a7314","resolution":{"observed_at":"2026-08-06T17:00:41.067248Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.872423Z","title":null,"venue":null,"work_id":"897d9be9-1a8a-426b-9043-d3b0ead865ee","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.761675Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:b0232ded1b8d6b653e9ae2aa36f1e1019774969d13dcb9a5925bb7ec961c8b5b","observation_id":"e48b24f3-0364-4ccd-8c9a-fde12479c565","resolution":{"observed_at":"2026-08-06T17:00:40.949596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.702370Z","title":"LQN25F020001), and in part by the Key R&D Program of Zhejiang (2025C01104)","venue":null,"work_id":"1db24f45-9621-413a-b7f4-dd599ca58a15","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.787020Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:72c5206c7bfb9d0855e74767c45f819e5dda2b411d677cbd0bb87a158629fb97","observation_id":"e559f109-d637-4a3a-879f-2e350a30d97a","resolution":{"observed_at":"2026-08-06T17:00:40.797255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.641592Z","title":"Naturalspeech: End-to-end text-to- speech synthesis with human-level quality,","venue":null,"work_id":"c1e3a7ca-a054-4aa9-9a40-a37d486caabf","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.344834Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:1e970ce8300dd680c51d1c9c366300e5e6871905f63d762676e5b7a74b344c1b","observation_id":"6a038c01-4ab0-470b-a154-038b0b632b2f","resolution":{"observed_at":"2026-08-06T17:00:39.711282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.523585Z","title":"Attention is all you need,","venue":null,"work_id":"9cc746de-d7fe-4047-891b-d1f57400f6f6","year":2017},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.844343Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:48fcab6034de38cd292b989995aa9823b972acd21884a8803d3b5862c50c76ff","observation_id":"6b0b10a4-0295-460e-83e8-27b08c43b8e7","resolution":{"observed_at":"2026-08-06T17:00:40.623965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.360298Z","title":"Fastspeech 2: Fast and high-quality end-to-end text to speech,","venue":null,"work_id":"3e5bfef4-a0f0-4c33-8d33-3c34cb2cdd41","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.970639Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:cc024e83aaca8f8e0c249a07c927a29a8acaa8f8b8322f12c97ecd3aa4af0013","observation_id":"86a78097-fab6-45f8-8473-3cf0aa740f55","resolution":{"observed_at":"2026-08-06T17:00:40.424716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.194686Z","title":"Glow-tts: A genera- tive flow for text-to-speech via monotonic alignment search,","venue":null,"work_id":"8d96ffa8-e20b-4e3c-9531-02bc326e766f","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.161679Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:20ccc28184fa9de2589e6413aaf010554ce4e5d99aaa3a312d9af4ba0d09f72a","observation_id":"1bb5d809-8fe0-4f8a-a5d7-4eb3638ff385","resolution":{"observed_at":"2026-08-06T17:00:40.274609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:40.001426Z","title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech,","venue":null,"work_id":"026311a3-34ca-43ad-9300-6397a053e54b","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.220293Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ddce1802a1e112f3f52f0e5ccad0e830480d68a781abf4669118acf34de288cb","observation_id":"9b58995a-8d18-4580-8fb0-eef9d33e24f1","resolution":{"observed_at":"2026-08-06T17:00:40.093806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.806308Z","title":"Grad-tts: A diffusion probabilistic model for text-to-speech,","venue":null,"work_id":"e7d7164f-4a4d-481b-9d3b-da987bc352b0","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.285892Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:46385e58e284b17961e76f643dff73d70e957b5e68421482a4284b4d1c77e71d","observation_id":"0e9bb990-4e1c-413b-9da5-0f0202d80a17","resolution":{"observed_at":"2026-08-06T17:00:39.897813Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.540675Z","title":"Msemotts: Multi-scale emotion transfer, prediction, and control for emotional speech synthesis,","venue":null,"work_id":"cbd92bf4-752f-40cb-a543-a4a1264fdd34","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.800700Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:b5311a4fa47b303e1e71ec7a23067bce92e3db5d61a7d5d4bca6296ddb384eca","observation_id":"aa80d873-4dbe-4f85-a15b-fa7f44ee11b1","resolution":{"observed_at":"2026-08-06T17:00:38.642831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.471655Z","title":null,"venue":null,"work_id":"39f29ece-3e6e-4962-a708-3b60f967d566","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.410247Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ab1090df274620bc040051c0742d59fa61aeeeebb5ee969c3958432dfcd09567","observation_id":"dddd2a01-6871-4070-9e94-e000cc52b04b","resolution":{"observed_at":"2026-08-06T17:00:39.558958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.267853Z","title":"Exploring transfer learning for low resource emotional tts,","venue":null,"work_id":"cc475391-433f-4873-9a1a-104cba0e0ff1","year":2019},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.487306Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:af269f2e642e897ae5d896c99b8339d8a483f9b6c2dd1db7e6b69039f60f5dc6","observation_id":"284d4e11-006c-4139-af63-d30e9848ab6e","resolution":{"observed_at":"2026-08-06T17:00:39.338794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:39.098919Z","title":"Emospeech: guiding fastspeech2 to- wards emotional text to speech,","venue":null,"work_id":"9424c13b-36cc-4549-8c1e-3c0a605cd10c","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.544569Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:59c16fa0ca1b2d1b19f2c4f1b43109e57b1aced5e44597b7ba9061c27e5851ff","observation_id":"ea7fac22-6097-4904-8b33-7671fa768928","resolution":{"observed_at":"2026-08-06T17:00:39.186750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.945669Z","title":"Emodiff: Intensity con- trollable emotional text-to-speech with soft-label guidance,","venue":null,"work_id":"a2da4601-ce32-46b5-94fe-d98263d84a00","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.650044Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ee025c75aa8f7da39fedbe3ac817e95dcf15ab451b4934edd78ac68574baead6","observation_id":"6bc51587-e293-4e72-9edf-f9b106b4b5e8","resolution":{"observed_at":"2026-08-06T17:00:39.018849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.768468Z","title":"Controllable emotion trans- fer for end-to-end speech synthesis,","venue":null,"work_id":"95c291b9-bbdc-4100-9153-18dcdc9745bd","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.743601Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:31652ced112a7fa178e422f529dbda2b5a153c68bdc5bc838c3a9706bcc1ec9c","observation_id":"be948db1-5421-4a71-8889-1a531c49bb20","resolution":{"observed_at":"2026-08-06T17:00:38.840631Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.434088Z","title":"Supervised and unsupervised approaches for controlling narrow lexical focus in sequence-to-sequence speech synthesis,","venue":null,"work_id":"00c04729-1e33-44f1-824c-e6bbe8c08284","year":2021},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.232683Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:eacd0c6f1d25bd3d284aa7a8e8b412dde97ca84528038c576bca48e29f03714c","observation_id":"f947fe3f-0bec-446f-8a11-b72a35a6b2bf","resolution":{"observed_at":"2026-08-06T17:00:37.484831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.362466Z","title":"Daft- exprt: Cross-speaker prosody transfer on any text for expressive speech synthesis,","venue":null,"work_id":"9700795d-0332-48eb-80a1-4ebf332bcd26","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.890240Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:52a09e57413c071fd541cc5a89d0a32ff11b6c10093ec8a10a9246eecfe29e79","observation_id":"ef779f2c-c39b-4297-9dd7-dec3972d390c","resolution":{"observed_at":"2026-08-06T17:00:38.455301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:38.119175Z","title":"The acoustics of word stress in en- glish as a function of stress level and speaking style,","venue":null,"work_id":"5aa02811-b276-40cb-9b6e-85556783c128","year":2015},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:33.967567Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:02ef506e1edec378a94d4f48d239ac2ebbe2d269a46416aff246b7ef8ab31e8a","observation_id":"be7b3e6e-a423-4945-8658-42504f261aa6","resolution":{"observed_at":"2026-08-06T17:00:38.272357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.932748Z","title":"The acoustics of lexical stress in italian as a func- tion of stress level and speaking style,","venue":null,"work_id":"f19cbe53-1da0-4077-9e8b-c15dc764f002","year":2016},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.027610Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:ff9f51a95f86553e2332b65122badcec2c7e2acadfce12fad30c7454b0061132","observation_id":"c911473b-ea8e-4858-b756-d0855fd19cd0","resolution":{"observed_at":"2026-08-06T17:00:38.036080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.782074Z","title":"Lex- ical stress perception as a function of acoustic properties and the native language of the listener,","venue":null,"work_id":"f175b854-fb61-4460-93e0-11895536ca02","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.084896Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:8252fe67f14aae327cc8bbdddf51bd974b8bbf5ac770c1ddc88a00e0d57b4849","observation_id":"a2e2b0c8-7fce-49fa-8e04-6ea76cd34714","resolution":{"observed_at":"2026-08-06T17:00:37.842467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.574689Z","title":"Emphatic speech generation with conditioned input layer and bidirectional lstms for expressive speech synthesis,","venue":null,"work_id":"3451eed2-fe38-4d8f-97a1-94940f2ac180","year":2018},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.167174Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:c48dd2b4bcfb0711a71e4d6051e5ebe3b63340a1f3ae93886c0353cabd73b036","observation_id":"1fca9010-44b4-40c8-89b5-f1aab7da3eb5","resolution":{"observed_at":"2026-08-06T17:00:37.669032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:41.126409Z","title":null,"venue":null,"work_id":"b0ffc8d8-920f-4882-b535-4998fb9f5b4d","year":null},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:32.662064Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:45c53619f825c8f07d655a3b33e0746557a58962d1915b12456ab475308f6b1e","observation_id":"79660d22-0c18-4033-a8fb-aab94c464fe3","resolution":{"observed_at":"2026-08-06T17:00:41.173964Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.253014Z","title":"Emphasis control for parallel neural tts,","venue":null,"work_id":"39f63422-f5d9-4259-b757-b25bba736d8a","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.332717Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:822c78338fe147cd17ba0b3d06f08137a740f5c302f7b2dfbe17ea74a5ecdfe0","observation_id":"259c828e-80bb-444b-b2a7-34632b3bd53e","resolution":{"observed_at":"2026-08-06T17:00:37.315296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:37.053796Z","title":"Ee-tts: Emphatic expressive tts with linguistic information,","venue":null,"work_id":"8449d7d5-961e-4506-8c3a-321ecd677147","year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.389567Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:4b1a6029546ed75b6e7c30cebcbf5fb82e0137c67760f9c2d61b551974324c9d","observation_id":"f819b3fa-7921-42ac-8baf-696418f5d8db","resolution":{"observed_at":"2026-08-06T17:00:37.146568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-14T10:40:26.323157Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-06T17:00:34.450317Z","title":"A survey of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.450317Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:d4725b5b3f1c6dc6a9ca6d1147d4926bad9ea9811a3c00d75618170c57f2c32f","observation_id":"eefe6eab-eba2-49eb-b8e6-393ef89e99f4","resolution":{"observed_at":"2026-08-06T17:00:34.450317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.856706Z","title":"Em- phAssess : a prosodic benchmark on assessing emphasis transfer in speech-to-speech models,","venue":null,"work_id":"7a0fbcb5-d371-4e29-b4ee-a9dc5ad39554","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.527568Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:17f53afec20bc1f04a858fefee5b2a55bdbeb3b2d93fbee2d512c7d1211a8124","observation_id":"b75f85c4-4b12-4115-8b86-c3b7c34fd809","resolution":{"observed_at":"2026-08-06T17:00:36.941557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:34.632284Z","title":"Emotional voice con- version: Theory, databases and esd,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.632284Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:214549fdafe6f1d21420fcaad6851fd52ca7efa0343362ab18987fe36dcafc84","observation_id":"2d4b70b6-65c2-4e28-b4b3-948dc72dbeb6","resolution":{"observed_at":"2026-08-06T17:00:34.632284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.624359Z","title":"Hierarchical rep- resentation and estimation of prosody using continuous wavelet transform,","venue":null,"work_id":"0cfd8b9e-653b-48f1-a74f-648fefdb40ac","year":2017},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.713790Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:2b0a4476e06d76c4d4d76bf502c67282ee725a5f8b55ffb5f14e99dbff5dd0c4","observation_id":"6e06e26b-7eab-48bf-924f-9882a5c7187b","resolution":{"observed_at":"2026-08-06T17:00:36.727689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.415987Z","title":"Adaspeech 4: Adaptive text to speech in zero-shot scenar- ios,","venue":null,"work_id":"20452868-c54b-4954-aa24-241ca41cfcfa","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.803675Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:9754fcf10b0826d77af41711831e0b77d119a542551eb345379b864d30ad1bcc","observation_id":"54819339-7cbf-4ad9-b4d7-a7ce54e7406f","resolution":{"observed_at":"2026-08-06T17:00:36.504645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.234021Z","title":"istftnet: Fast and lightweight mel-spectrogram vocoder incorporating inverse short-time fourier transform,","venue":null,"work_id":"4469679c-9fc9-4da2-a0e5-b46926eaffe0","year":2022},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.887690Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:8013ee3ae1423a34bbf4981e0115d8e46367076ad70641fb37a2b23c9a2b31d7","observation_id":"ea73529f-bfe4-47af-a1af-711f4ba6092e","resolution":{"observed_at":"2026-08-06T17:00:36.326584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:36.097689Z","title":"Adam: A method for stochastic optimization,","venue":null,"work_id":"96acfd5b-8cf8-450c-8dbc-dccc0d7752b9","year":2015},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:34.970839Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:7f7dfe2a205fd589242d17d633d7be9d0ba5c0ee7fed5ff4ff8eb824d604db7b","observation_id":"2acf7453-1155-4e6c-b3e4-534d2479583e","resolution":{"observed_at":"2026-08-06T17:00:36.152787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-16T06:25:22.037199Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-06T17:00:35.058089Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.058089Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:52e57c1a9af80e237bf8e2d4dbf0763e45e6a5e0ebe8b9c496ce11b238a4f47d","observation_id":"9df9ccfb-e94f-4307-b1ac-b3977703d489","resolution":{"observed_at":"2026-08-06T17:00:35.058089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T17:00:35.146817Z","title":"GPT-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.146817Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:2f8aad0f90c9dee72b2cba1206fed914636edb221b21fd73e7c2ded8a737efd9","observation_id":"d419d07b-6e0d-42ae-aee4-43e7d537a330","resolution":{"observed_at":"2026-08-06T17:00:35.146817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:35.911986Z","title":"emotion2vec: Self-supervised pre-training for speech emotion representation,","venue":null,"work_id":"78a2d100-f07c-47b1-8916-84538949e125","year":2024},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.245582Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:49a0fcb72043804de09d57359bc954d3d7e21844e4fc618766a6fcc9b8a2eb4d","observation_id":"9a5cd7ae-cd45-4b15-a515-db92184589a8","resolution":{"observed_at":"2026-08-06T17:00:36.008196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T17:00:35.706854Z","title":"Deep learning based assessment of syn- thetic speech naturalness,","venue":null,"work_id":"55602d7b-acd2-4307-9da9-fcc3727fc06f","year":2020},"citing_paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T17:00:35.341553Z"},"links":{"citing_paper":"/paper/2507.12015"},"observation_digest":"sha256:fd117349076dca258fdb924911d8b0ac3cc377b1c4e55e6f233fd2f653865ccd","observation_id":"3c4a8d21-99e2-4594-b225-9a385ed9ef18","resolution":{"observed_at":"2026-08-06T17:00:35.809221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.12015","last_updated":"2025-07-16T08:19:20Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-06T16:53:37.184878Z","submitted_at":"2025-07-16T08:19:20Z","title":"EME-TTS: Unlocking the Emphasis and Emotion Link in Speech Synthesis"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":8,"verified_exact":0,"verified_fuzzy":28},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 1 inbound Pith citation observation for arXiv:2507.12015."}