{"as_of":"2026-08-19T11:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5c9ce7317efcef798b3d26053c0831a3bb0207eab08a149d3ecacce37f626816","coverage":[{"denominator":42,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":42,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:29:43.405193Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.04904/citation-record","integrity":"/paper/2501.04904/integrity","json":"/paper/2501.04904/citation-record.json","paper":"/paper/2501.04904"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.874239Z","title":"Natural TTS Synthesis by Conditioning Wavenet on Mel Spectrogram Predictions,","venue":null,"work_id":"98475047-87cc-4d15-8a3a-3a4361b52af9","year":2018},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.250452Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:7f848b78a0e1ffd5f6f2ed29ee64a05f718780c8681d76d61b8e3666c538c3f5","observation_id":"455ed74b-dcef-479c-9e4e-35ffe4ac91b9","resolution":{"observed_at":"2026-08-10T21:29:43.878455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.862268Z","title":"FastSpeech 2: Fast and High-Quality End-to-End Text to Speech,","venue":null,"work_id":"edd3d284-e07a-4534-9c62-153a67a060c7","year":2021},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.254928Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:38acde89ce211a43ad1f339b44b3007d8c7f64b241e10efc3ab3bed9d62deee9","observation_id":"5775c4de-2f6f-4eb9-989d-47f62652191c","resolution":{"observed_at":"2026-08-10T21:29:43.866233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.849404Z","title":"HierSpeech: Bridging the Gap between Text and Speech by Hierarchical Variational Inference using Self-supervised Representations for Speech Synthesis,","venue":null,"work_id":"38ce3464-99fe-4a8e-b7b6-c19c06e20ae5","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.258540Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:9f7aed0757f2bbb50bbeb8cb3e4fd4ff105b3ebb811ba4c56fa7ea6d0f9f0d4f","observation_id":"e7e85684-bfa0-4440-a5d5-4eb2e7cadf91","resolution":{"observed_at":"2026-08-10T21:29:43.853674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.837122Z","title":"Matcha-TTS: A fast TTS Architecture with Conditional Flow Matching,","venue":null,"work_id":"f24a0155-8418-46eb-a834-936a0db3da52","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.262268Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:672fdff3aaa42594a097e82c1b5021a17e30a5a3e82312c202d8fe20a4d9e8b2","observation_id":"90b748e9-fa3a-4628-835d-8b599e50947f","resolution":{"observed_at":"2026-08-10T21:29:43.841512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.825490Z","title":"V oice- Flow: Efficient Text-To-Speech with Rectified Flow Matching,","venue":null,"work_id":"41a03475-dd07-45ff-b195-de4531894259","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.266357Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a6f53f4a31e0d7cf0328375dd82c59f07b78a722a77603987d14f683569b4437","observation_id":"e20bbb16-47d6-40d3-ba20-4e328af25522","resolution":{"observed_at":"2026-08-10T21:29:43.829346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.812850Z","title":"Conversational End-to-End TTS for V oice Agents,","venue":null,"work_id":"1fbf7a84-b7de-494e-ae7e-bed59e0773de","year":2021},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.269897Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a14726a64c2cb562a995b18788ea34414ca4464594ff71304ae815286868aa5d","observation_id":"32510b16-9547-40a3-8b5f-d1a683c714f9","resolution":{"observed_at":"2026-08-10T21:29:43.817053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.799983Z","title":"Enhancing Speaking Styles in Conversational Text-to- Speech Synthesis with Graph-Based Multi-Modal Context Modeling,","venue":null,"work_id":"e62ee6c2-40d0-4e97-999d-555e9ad60f3a","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.273627Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:24152563cfc297f7bb5b21a3d51abbce21667234a7a6b7d3877b336f633ca2b9","observation_id":"ba724b39-767d-444f-9a78-f4f13c82626b","resolution":{"observed_at":"2026-08-10T21:29:43.804447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.786929Z","title":"M2-CTTS: End-to-End Multi-Scale Multi-Modal Conversational Text-to-Speech Synthesis,","venue":null,"work_id":"57290c54-3797-4b78-a391-e779c77e4c41","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.276720Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:2477c407eddaf745127d48fcf239a3cebb2237ed70df11c6813e42700aee33ad","observation_id":"ee3e98f9-7070-44e0-931e-9431589ab592","resolution":{"observed_at":"2026-08-10T21:29:43.791811Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.777108Z","title":"Concss: Contrastive-based Context Comprehension for Dialogue-Appropriate Prosody in Conversational Speech Synthesis,","venue":null,"work_id":"a9a3c139-72a3-42d1-9a0a-a3da71d4b7d4","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.280053Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:25130e64cf8d35e974442d189894757ff74ad37e2d3fe83cd5a6b75a735b828a","observation_id":"742097bf-bc69-4f58-8f98-1d2a851855ec","resolution":{"observed_at":"2026-08-10T21:29:43.780401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.768554Z","title":"Considering Temporal Connection between Turns for Conversational Speech Synthesis,","venue":null,"work_id":"6be81d14-5dc2-4b13-b011-f5871136a5c1","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.283619Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:64417b26fc65590c0ef40303d927a6c8a099da1ca4267b624d0089023b3182ea","observation_id":"5ce4b5e3-cffa-4fc9-9454-d64d312ae447","resolution":{"observed_at":"2026-08-10T21:29:43.771439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.758495Z","title":"A new recurrent neural-network architecture for visual pattern recognition,","venue":null,"work_id":"1ad02544-f9fe-4ef8-8450-5be2c760bdf3","year":1997},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.287141Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:539375735c31b15adca0447c1d6d0c73553395a3e2997cee5e63e818cc142481","observation_id":"fb2c1b8d-15ba-40ce-bab2-c8de5594d452","resolution":{"observed_at":"2026-08-10T21:29:43.762313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.747807Z","title":"Classification of drowsiness levels based on a deep spatio-temporal convolutional bidirectional LSTM network using electroencephalogra- phy signals,","venue":null,"work_id":"b779ae72-eac0-47b4-ae37-289211a719d4","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.290406Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:5bebf406536d01160999a1d009dc555d9af37bc9a92cd1cc673713626e9e8d76","observation_id":"3a40749b-a6f4-4121-b242-d43c29bd8288","resolution":{"observed_at":"2026-08-10T21:29:43.751693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.737099Z","title":"Towards an EEG-based intuitive BCI communication system using imagined speech and visual imagery,","venue":null,"work_id":"d5e85884-baae-4faa-a197-e6db406063ab","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.293873Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:e0785c877a6be1226448ee1cbb6ab1adf920690a76c6021b0931d2bde3d5111d","observation_id":"beffdf87-2a96-4cb6-8721-f96253610690","resolution":{"observed_at":"2026-08-10T21:29:43.741084Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.724560Z","title":"A multi-view cnn with novel variance layer for motor imagery brain computer interface,","venue":null,"work_id":"4a4b91ad-79ad-43d0-91f9-3ca11810bc25","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.297415Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:b51765df9e6c102c6d5c4efec27eef0198266f2ed38b1cee23e40b76439ce5fe","observation_id":"164e2906-c334-4023-94c1-e87a15e31658","resolution":{"observed_at":"2026-08-10T21:29:43.729368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.713972Z","title":"An adaptive deep reinforcement learning framework enables curling robots with human-like performance in real-world conditions,","venue":null,"work_id":"67f7eb1d-e164-4c72-a242-76b1e4f041fc","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.300931Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0f712c9493b0f3af04ec7cdb6fd4be09ad6ccdae011152850cda48b49c27be32","observation_id":"fcafa22c-1ec9-4a12-a8e9-a0e61124b666","resolution":{"observed_at":"2026-08-10T21:29:43.717871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07547","last_updated":"2024-08-14T13:36:17Z","snapshot_observed_at":"2026-08-16T13:26:29.252459Z","submitted_at":"2024-08-14T13:36:17Z","title":"PeriodWave: Multi-Period Flow Matching for High-Fidelity Waveform Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07547","snapshot_observed_at":"2026-08-10T21:29:43.304427Z","title":"Periodwave: Multi-period flow matching for high-fidelity waveform generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.304427Z"},"links":{"cited_paper":"/paper/2408.07547","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0409da42955e3af3fdb9c8cc60d0c52f36ce1a43fe32582cf2849fa7a9343f11","observation_id":"9c98865d-edeb-494d-960a-ebc03960a0c9","resolution":{"observed_at":"2026-08-10T21:29:43.304427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.703606Z","title":"Emoq-tts: Emotion intensity quantization for fine-grained controllable emotional text-to-speech,","venue":null,"work_id":"c324eefe-a835-4ede-bced-509408cb3f9b","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.308697Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:94eae281e77e7c7d964ac3539ef3875d0bd776e3bae2dd2744bd78dc73090379","observation_id":"6ab269ed-6f19-44d1-88c0-899b507cf3f8","resolution":{"observed_at":"2026-08-10T21:29:43.707472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.692794Z","title":"Diffprosody: Diffusion-based latent prosody generation for expressive speech synthe- sis with prosody conditional adversarial training,","venue":null,"work_id":"41783aa6-25fd-4047-b455-23af23f4f08f","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.312310Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:346eb24958df51d1c7730a737e8772ccc5dea0f1fdaa8edef1846cbe4ef8760a","observation_id":"6996cda2-e285-40fb-830d-fb887bd96225","resolution":{"observed_at":"2026-08-10T21:29:43.696860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.681980Z","title":"EmoSphere-TTS: Emotional Style and Intensity Modeling via Spherical Emotion Vector for Controllable Emotional Text-to-Speech,","venue":null,"work_id":"c82858ff-be03-4f47-9c92-c7eaecaac4e1","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.316220Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:d4c55f5eda35e2420bda417f33d2d9422fb6de1e9e8a7069b4018b9dbdaf4987","observation_id":"984f69b8-44a3-4d20-8b4a-8b5c772f9de0","resolution":{"observed_at":"2026-08-10T21:29:43.686283Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.08095","last_updated":"2025-01-21T02:51:53Z","snapshot_observed_at":"2026-08-18T08:00:12.550924Z","submitted_at":"2024-01-16T03:39:35Z","title":"DurFlex-EVC: Duration-Flexible Emotional Voice Conversion Leveraging Discrete Representations without Text Alignment","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.08095","snapshot_observed_at":"2026-08-10T21:29:43.320033Z","title":"DurFlex-EVC: Duration-Flexible Emotional V oice Conversion with Parallel Generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.320033Z"},"links":{"cited_paper":"/paper/2401.08095","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:bfc1929837dae8ff845c56501b9032a8a8c5ade5f3c28f1a8c896fdfb184d751","observation_id":"341a8fea-0dac-4930-8223-911bc5be0baf","resolution":{"observed_at":"2026-08-10T21:29:43.320033Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.671968Z","title":"Emotion rendering for conversational speech synthesis with heterogeneous graph- based context modeling,","venue":null,"work_id":"28cd1abc-d0ab-4bae-9403-e547608cb909","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.324595Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:383b427d7dc23d056c085e5663c11528b3bde9af12fb83ba9ed5f9a19db5b5c0","observation_id":"94e8086e-f874-41c0-a17f-f6512c75925b","resolution":{"observed_at":"2026-08-10T21:29:43.675754Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-16T05:38:56.513422Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16420","snapshot_observed_at":"2026-08-10T21:29:43.328351Z","title":"Internlm-xcomposer2: Mastering free-form text- image composition and comprehension in vision-language large model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.328351Z"},"links":{"cited_paper":"/paper/2401.16420","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:e88a127787529b20c95f7678b23852053144bf7b7308288448d3fad90bd095ae","observation_id":"7e155f12-d29e-4db2-ac40-b730b9212873","resolution":{"observed_at":"2026-08-10T21:29:43.328351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.19041","last_updated":"2024-05-29T12:32:08Z","snapshot_observed_at":"2026-08-17T16:07:59.879274Z","submitted_at":"2024-05-29T12:32:08Z","title":"BLSP-KD: Bootstrapping Language-Speech Pre-training via Knowledge Distillation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.19041","snapshot_observed_at":"2026-08-10T21:29:43.332430Z","title":"BLSP-KD: Bootstrapping Language-Speech Pre-training via Knowl- edge Distillation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.332430Z"},"links":{"cited_paper":"/paper/2405.19041","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:6b91102fa96ee7b92d1d27fcd50d30546ab56b708e31a619f8c6365edad4b4cc","observation_id":"d4b4056f-d47d-4147-a27a-19bb8f70547a","resolution":{"observed_at":"2026-08-10T21:29:43.332430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.661126Z","title":"Whisper-AT: Noise-Robust Automatic Speech Recognizers are Also Strong General Audio Event Taggers,","venue":null,"work_id":"95eb563a-d53a-4b29-b987-a86b24a20274","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.336305Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:61124ca0ce84d0e56a2ebb00df48aeda73909aabc09e360d4bd9540987442eb1","observation_id":"6e31169c-973e-422b-8cda-896d7e88e3e7","resolution":{"observed_at":"2026-08-10T21:29:43.665088Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.651661Z","title":"BLIP-2: Boot- strapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models,","venue":null,"work_id":"72db5829-77da-41e5-bf77-d6e4dc853c05","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.340004Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:5637caa0f60b4662b035291f9a85afcfd062fae7e36e1d7cea5e2e4d90ca8364","observation_id":"72786add-657c-47f6-a067-d0c711000053","resolution":{"observed_at":"2026-08-10T21:29:43.654723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.643098Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision,","venue":null,"work_id":"b2ff37b1-3066-4652-b7a0-73f7ca42d195","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.343722Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0a2e1aeca6753b129cc71d1b8b99fbb72168581d8ec2c968437cac9a8b3b60ac","observation_id":"d6b51203-c677-478f-bc65-b3d7ff262897","resolution":{"observed_at":"2026-08-10T21:29:43.646248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.634437Z","title":"Joint Audio and Speech Understanding,","venue":null,"work_id":"bcb082b4-66dc-4966-ba69-a68faaeba9d8","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.347055Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:edff95beb19d2341549846cf01131ebe007347e707052f516de7998c281d0bcc","observation_id":"2b3da1c3-3ed5-40f3-b68a-f60507af2742","resolution":{"observed_at":"2026-08-10T21:29:43.637554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.623884Z","title":"HiFi-GAN: Gen- erative Adversarial Networks for Efficient and High Fidelity Speech Synthesis,","venue":null,"work_id":"a1427cc5-5d26-4413-8ed2-554c4f14b038","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.350771Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:c01c291498eecbeb7ec47baafd4b170944b5d8b56c671234c454ae9bc60f84ef","observation_id":"4677b17d-40d3-4301-8b47-3c7cf683dc18","resolution":{"observed_at":"2026-08-10T21:29:43.627949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.613427Z","title":"DailyTalk: Spoken Dialogue Dataset for Conversational Text-to-Speech,","venue":null,"work_id":"ccc6e73a-67b8-4bec-adff-8ac96edd5c4c","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.354260Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:68307122f41fff3cbd612e97aa02ae26682934ca57c006ccf2e0bc99cd39c851","observation_id":"147b7bcf-18f5-4439-b418-170a3605770e","resolution":{"observed_at":"2026-08-10T21:29:43.617661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.603042Z","title":"DailyDialog: A Manually Labelled Multi-turn Dialogue Dataset,","venue":null,"work_id":"5cc177db-7b27-44b8-88b9-4a5ccab1fa62","year":2017},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.357664Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:4d791b3c84f5ff020b8b6421b57fe7117aada05fc36039dfed196dc038bccf1f","observation_id":"f0ae0780-a0db-4eb4-a9f3-39cd9a8dff17","resolution":{"observed_at":"2026-08-10T21:29:43.607229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.592353Z","title":"CREMA-D: Crowd-Sourced Emotional Multimodal Actors Dataset,","venue":null,"work_id":"189872cb-0f17-4740-96e1-862e144955b9","year":2014},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.361632Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:0628c517567a2197538bcc290bc837f281ffd5e12114f0d6a7cc5912e4999155","observation_id":"52c67a9b-2bec-4a87-9803-ea1867d58737","resolution":{"observed_at":"2026-08-10T21:29:43.596339Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1806.09514","last_updated":"2018-06-25T15:01:54Z","snapshot_observed_at":"2026-08-14T18:59:34.605079Z","submitted_at":"2018-06-25T15:01:54Z","title":"The Emotional Voices Database: Towards Controlling the Emotion Dimension in Voice Generation Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.09514","snapshot_observed_at":"2026-08-10T21:29:43.365529Z","title":"The emotional voices database: Towards controlling the emotion dimension in voice generation systems,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.365529Z"},"links":{"cited_paper":"/paper/1806.09514","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:150a3da0a1092d4897bda59c033a80dd6737f0fb2acc372f418efcf6d57677d7","observation_id":"826908ef-0d70-4fe4-9dcb-0992498ea9d1","resolution":{"observed_at":"2026-08-10T21:29:43.365529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.580829Z","title":"IEMOCAP: Interactive emotional dyadic motion capture database,","venue":null,"work_id":"b207d8b2-f048-48a2-9f0e-47293c6e052d","year":2008},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.369873Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:69df71bda0138c3a64db91b8ea17f40719517bfbd8c36321db3c94b9a963f4ac","observation_id":"641c9827-6c87-4460-9c1b-7c46aef6d3cc","resolution":{"observed_at":"2026-08-10T21:29:43.585460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.571082Z","title":"MEAD: A Large-scale Audio-visual Dataset for Emotional Talking-face Generation,","venue":null,"work_id":"a80fc278-e88e-455b-93eb-aab28cefa249","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.373630Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:15a190efc9e6c6398567fcbb208d4b46649630b8fd8894165953c06bb50bce9d","observation_id":"7171c184-92f1-43ca-8ef0-bf1fc0f118e5","resolution":{"observed_at":"2026-08-10T21:29:43.574845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.560055Z","title":"Toronto emotional speech set (tess)-younger talker happy,","venue":null,"work_id":"113f0e51-ff71-4b46-8159-c8fbcbcf30c3","year":2010},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.377953Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:ecaf958770cec1eceb8ec8981cdde010d83ce1e64184154623fb14b10b3b5c13","observation_id":"6c25328d-a94d-4d36-a1fe-da76096e92e1","resolution":{"observed_at":"2026-08-10T21:29:43.564236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.548897Z","title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality,","venue":null,"work_id":"04d1dcf3-8de1-4f46-895e-7bd7125d40b4","year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.381638Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a811be3f01f22746b0efd6da728008711fe8972963d730556d54e9b1e554bdea","observation_id":"2c0d3525-bab5-4ed6-b275-f2e7825b8c83","resolution":{"observed_at":"2026-08-10T21:29:43.552689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-10T21:29:43.385693Z","title":"LLama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.385693Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:f0c0d3a4380379e9ab3eb0355445b3ae940c09b3e510a0ce28124a83ba4c9dff","observation_id":"4def9cc2-dc2c-448f-ae21-4702a0626c30","resolution":{"observed_at":"2026-08-10T21:29:43.385693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.538699Z","title":"Decoupled Weight Decay Regular- ization,","venue":null,"work_id":"4de78178-11d7-4b9b-b382-52e5eb80bfc3","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.389703Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a1d41785eccf3dfe0ee201059d976e100ec3cafb47790a6fd890587515cbca29","observation_id":"7648c2f8-f3da-4d9d-ae90-fab86882c09a","resolution":{"observed_at":"2026-08-10T21:29:43.542313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.528162Z","title":"emotion2vec: Self-Supervised Pre-Training for Speech Emotion Representation,","venue":null,"work_id":"3f050773-dced-4b67-b571-dbcf46034b8a","year":2024},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.393629Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:420968abe379eb80ec6cdbc421718ad379e4cfab06ab2c1b99ed0a594fb24d9d","observation_id":"67b43322-bc93-41fb-b1cf-9886ae39087c","resolution":{"observed_at":"2026-08-10T21:29:43.531861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.517821Z","title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations,","venue":null,"work_id":"222d9879-1fae-49e6-9ec2-04e6d440e976","year":2020},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.397315Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:a8a1d9b1e1da673bf21bd0d35eefa4b5ce4bc666abb40abe7dca790940dd8f77","observation_id":"5f18c2de-8774-4623-9571-2f397f39f659","resolution":{"observed_at":"2026-08-10T21:29:43.521601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.507564Z","title":"Sequence-to-Sequence Acoustic Modeling for V oice Conversion,","venue":null,"work_id":"0acf2324-0bf5-4c79-8a6c-55c947015f2c","year":2019},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.401154Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:8cd55b274c990a183a6d1312eec30c650068027babcb8b452e25f98d4f687256","observation_id":"9c39e78d-c66c-41b2-9c47-f32b48e68740","resolution":{"observed_at":"2026-08-10T21:29:43.511085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:29:43.494193Z","title":"LoRA: Low-Rank Adaptation of Large Language Models,","venue":null,"work_id":"65556e09-ef71-407a-993f-8fbae8fb6eba","year":2022},"citing_paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-10T21:29:43.405193Z"},"links":{"citing_paper":"/paper/2501.04904"},"observation_digest":"sha256:29d27c4efbc36776a2aa935a410b0944dd91b5a2528f874a618dcafa9d876fcb","observation_id":"4df5316b-c29f-4c04-9415-635dde855aaa","resolution":{"observed_at":"2026-08-10T21:29:43.499432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.04904","last_updated":"2025-01-09T01:32:44Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-15T10:41:30.424361Z","submitted_at":"2025-01-09T01:32:44Z","title":"JELLY: Joint Emotion Recognition and Context Reasoning with LLMs for Conversational Speech Synthesis"},"reference_resolution":{"displayed":42,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":6,"verified_exact":0,"verified_fuzzy":36},"total_outbound_references":42},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 42 of 42 outbound references and 0 inbound Pith citation observations for arXiv:2501.04904."}