{"as_of":"2026-08-23T17:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c9f1c0c1667b1a78c4c0477ce1453f3602e9ded05d0a0f12c002fb6b43624c2c","coverage":[{"denominator":47,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":47,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T19:24:43.629693Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.16716/citation-record","integrity":"/paper/2506.16716/integrity","json":"/paper/2506.16716/citation-record.json","paper":"/paper/2506.16716"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.458660Z","title":"Video Summarization Using Deep Neural Networks: A Survey,","venue":null,"work_id":"3010948f-7a4a-4c13-b409-4386e9cacc79","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.394542Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:fb47022e60264c77e0d0b72a53e258aab691cea8f43225dad0aa2e14fd81e337","observation_id":"b8860e63-d9d5-451c-97d2-c2be9251fdfb","resolution":{"observed_at":"2026-08-15T19:24:44.464110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.441118Z","title":"Exploring Video Captioning Techniques: A Comprehensive Survey on Deep Learning Methods,","venue":null,"work_id":"380e6a69-05ef-4945-ad6b-82aef385d6f1","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.400358Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:5e91375289cdc1fcb6f8fe2c91e55eaf3be76868e91546b58ec7ca4366fc5489","observation_id":"158d775e-5328-4e1a-9c82-27ee2e104a6e","resolution":{"observed_at":"2026-08-15T19:24:44.446713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.423947Z","title":"Automatic Image and Video Caption Generation With Deep Learning: A Concise Review and Algorithmic Overlap,","venue":null,"work_id":"a444e3f7-0b54-4f9f-8fd4-03453f47284f","year":2020},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.405355Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:8c3942131656f752342def5319dc241e3e592fc4c25cd0908a41f84b5a0ac88e","observation_id":"406f83ab-514b-45e2-bf0b-beeb70fe0fa2","resolution":{"observed_at":"2026-08-15T19:24:44.428989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.407766Z","title":"Paralinguistic and spectral feature extraction for speech emotion classification using machine learning techniques,","venue":null,"work_id":"1d6e7fce-0a5e-4e2f-8645-08bf5887e628","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.410147Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:aef6d9533e77b0ecce64bfa8cfc3191351893d96eb0bf099ce9669f415a06682","observation_id":"e9c05887-a454-4d89-a9d9-738dfb89e2c1","resolution":{"observed_at":"2026-08-15T19:24:44.412894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.391490Z","title":"V ocal communication of emotion: A review of research paradigms,","venue":null,"work_id":"fe94cbfb-8060-4bd1-9a85-d6b903b9877b","year":2003},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.415413Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:90529988b588cb088957e450e0c92757c8f485e85303913290408bc1d6cbcbc9","observation_id":"2a5d50ef-e210-4feb-9889-4cbb998f71bb","resolution":{"observed_at":"2026-08-15T19:24:44.396586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.375202Z","title":"Does speech rate influence intertemporal decisions? an experimental investigation,","venue":null,"work_id":"ed55b1fe-298d-4354-b227-cc7badc29d0a","year":2022},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.420385Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:0085bc892fb1fc163ac38f3a7d6f28b16b6141c09e90fe0b3d8428cfcfc60caa","observation_id":"01bd08e2-d22d-40e3-a44d-0acfa3074792","resolution":{"observed_at":"2026-08-15T19:24:44.380378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.359055Z","title":"Rhythmic and speech rate effects in the perception of durational cues,","venue":null,"work_id":"7884b1da-d8c4-4856-a880-67f4885a5f46","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.426198Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:60bb6986dfe5a2f42b9d02c518d149c08e08283f5d2479cc882e7a21f0c47402","observation_id":"7ae59257-e3a4-467e-acef-53e88500a753","resolution":{"observed_at":"2026-08-15T19:24:44.364329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.342151Z","title":"Emotion and Motivation,","venue":null,"work_id":"fda1fc8b-6a93-4b4e-8725-be7e06abaa41","year":2007},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.431109Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:beb70bab25532ae60c613e2a3e0ad51cd8930f5d54890a4d56f3ec02e64e327d","observation_id":"a2ade0d4-3000-4419-baed-af51b68cd87e","resolution":{"observed_at":"2026-08-15T19:24:44.347065Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.325347Z","title":"Language and Emotion: Introduction to the Special Issue,","venue":null,"work_id":"c998dd7f-0cdf-400a-957d-466915839627","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.435971Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:add361f429e29384961cf98ac6ec7608c7dc0410a29df53d02c537682209411b","observation_id":"1206c63c-2edc-40d9-abe7-de2313d37242","resolution":{"observed_at":"2026-08-15T19:24:44.330583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.308404Z","title":"Audio Description Generation in the Era of LLMs and VLMs: A Review of Transferable Generative AI Technologies,","venue":null,"work_id":"3fe32822-9e99-46a0-8448-2e571074968f","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.440950Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:379aa8632083fd4d4b1a835795b2f6ba5ed3749f38971e156dfdd656c6d9fd0a","observation_id":"eb766553-021e-4fd9-baef-4af8c773e48d","resolution":{"observed_at":"2026-08-15T19:24:44.313997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.291256Z","title":"Audio Description in the UK: What works, what doesn’t, and understanding the need for personalising access,","venue":null,"work_id":"9728a1cd-aea1-4a2a-95bc-deb0cf9c1f40","year":2018},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.445798Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:091a09b6902392383e2cd0fb0012ed4823eec15408a2beae9ac0196ae6823435","observation_id":"154ff93f-b5fc-4b23-800e-19f63f840a1a","resolution":{"observed_at":"2026-08-15T19:24:44.296960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.273264Z","title":"Audio description: The visual made verbal,","venue":null,"work_id":"a730fbcf-4d85-473a-9e44-41fb4c3c1665","year":2005},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.450609Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:08d1de770bb635918f8a144a8f55905c4850a9f37cdf28acdb93f76fb6b7e177","observation_id":"e242ada3-45d2-4b6e-a2de-bea2dfc8a49d","resolution":{"observed_at":"2026-08-15T19:24:44.279056Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.254646Z","title":"Ambient Lights Influence Perception and Decision-Making,","venue":null,"work_id":"23bd170e-db6b-4953-8983-60bd3b0ea6cc","year":2019},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.455802Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:7341d69edf8389606050419161f6117f4ca8a9c68149f06ebcaaa76cdcd5ba03","observation_id":"7ae61579-eaf9-49c0-836d-540cf762688e","resolution":{"observed_at":"2026-08-15T19:24:44.260231Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.237775Z","title":"Kobayasi,Colorist: a practical handbook for personal and profes- sional use","venue":null,"work_id":"46051996-c004-4d16-8d9a-67918e11e877","year":1998},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.463527Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:9f5afabd351c020ea8d67f55821346fa5d293ed3f44964710f8c650f955b1e8f","observation_id":"54bb42c2-df1a-4170-baae-cb96245cc62e","resolution":{"observed_at":"2026-08-15T19:24:44.243014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.220847Z","title":"Bordwell and K","venue":null,"work_id":"a83c40dc-fac5-4d71-aadd-46c077331e4d","year":2013},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.468534Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:fb92cc80a8419b00ee87dff07fed7c30b5624a8d058093325ca41639d0b5ebef","observation_id":"da44f779-e6a1-4dc9-b894-433c89c40ec4","resolution":{"observed_at":"2026-08-15T19:24:44.226099Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1703.10135","last_updated":"2017-04-06T21:20:34Z","snapshot_observed_at":"2026-08-17T12:01:30.659594Z","submitted_at":"2017-03-29T16:55:13Z","title":"Tacotron: Towards End-to-End Speech Synthesis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1703.10135","snapshot_observed_at":"2026-08-15T19:24:43.473186Z","title":"Tacotron: Towards end-to-end speech synthesis,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.473186Z"},"links":{"cited_paper":"/paper/1703.10135","citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:01773319339ace841e0495ef84e258f76e9028cab2498341a3f7bec16eb1d48a","observation_id":"30296089-67be-4a4f-b0a7-9c35ccd29167","resolution":{"observed_at":"2026-08-15T19:24:43.473186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.203755Z","title":"WaveNet: A Gener- ative Model for Raw Audio,","venue":null,"work_id":"400bec3e-8dd7-492a-880b-50de30d356c2","year":2016},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.478474Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:fc409495b91f8dce16234151857bc5b2d04f29388af6f127f32e0b7183173dee","observation_id":"cc9c4cc1-59e0-4f63-a266-cf91ff424250","resolution":{"observed_at":"2026-08-15T19:24:44.209238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.186887Z","title":"Styletts 2: Towards human-level text-to-speech through style diffusion and adversarial training with large speech language models,","venue":null,"work_id":"c0d43ee1-b7df-4fab-ad48-9fd924465d1b","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.483267Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:3a37d5dcbfa6417b4e05c862209ee53639f87ed6fa261a12bb82ab0717198646","observation_id":"8d6dbc91-1d9d-4bcf-b033-7731b722fa84","resolution":{"observed_at":"2026-08-15T19:24:44.192402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.488222Z","title":"Yourtts: Towards zero-shot multi-speaker tts and zero-shot voice conversion for everyone,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.488222Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:b73ce132a29adadc6d23828ca57b256100d5e371c7164ac7bc6fd23150b3d8ae","observation_id":"9fc1bf72-70ac-4baa-8d65-5f4eb52cfbc1","resolution":{"observed_at":"2026-08-15T19:24:43.488222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.158420Z","title":"InstructTTS: Modelling Expressive TTS in Discrete Latent Space with Natural Language Style Prompt,","venue":null,"work_id":"6f4bd085-1dd4-4bdb-9691-1e0ddca54ee5","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.492914Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:1b1b62308a92e94b45e5eb7bbbd1d9b8c4ed6d96d36c45af9e043a0e1cfdbc97","observation_id":"f5c9a6db-24f0-45a6-b00d-10560b18e533","resolution":{"observed_at":"2026-08-15T19:24:44.163673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.141913Z","title":"V oxinstruct: Expressive human instruction-to-speech generation with unified multilingual codec language modelling,","venue":null,"work_id":"1156ab66-8b75-47be-9afe-ddc551706f6f","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.498094Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:3e439758ef4f3e4fcf553e97e1d0378ca00c4702a6cee5a796cb2298ce4bda12","observation_id":"1a85e798-4907-4c7d-9f46-97d327645299","resolution":{"observed_at":"2026-08-15T19:24:44.147043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.124924Z","title":"TextrolSpeech: A Text Style Control Speech Corpus with Codec Language Text-to-Speech Models,","venue":null,"work_id":"1a1140c0-b91d-4837-be07-3fb8d79a34ff","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.503050Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:51c6d6ceaa317df4dfd7cba7033167cca5ddd70b45692fc60cfe9fec5d8f549c","observation_id":"fbf54452-192a-4421-8cd2-3377ceb8c5af","resolution":{"observed_at":"2026-08-15T19:24:44.130230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.107875Z","title":"PromptTTS 2: Describing and Generating V oices with Text Prompt,","venue":null,"work_id":"c1396ed0-9bc8-4b46-a972-bf486e23160f","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.507944Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:f9fa461c6e32d174615f0bc7d8cc99014d08384b1896db7bcb5eacbbd3ff61e4","observation_id":"d6eec240-f6d7-4dea-b8db-02a3be46a599","resolution":{"observed_at":"2026-08-15T19:24:44.113137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.090896Z","title":"What Does Your Face Sound Like? 3D Face Shape towards V oice,","venue":null,"work_id":"08bc6d09-d67f-4bfa-a006-52e4eb502872","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.512894Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:ddc160ad258c957a6214d235e95ea95aed3e7c8a8290dd27fa852eb9d8bcd527","observation_id":"574dda41-3981-4538-892a-91d362b3df7f","resolution":{"observed_at":"2026-08-15T19:24:44.096453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.072438Z","title":"Multimodal Machine Learning: A Survey and Taxonomy,","venue":null,"work_id":"00ea080c-98f4-4e26-833e-af44cecda8ef","year":2019},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.517703Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:69c2a12c418e1ec17c2227f91eb54710ba0a35b308eef77bac71fb06375a8cc0","observation_id":"05f92937-b0e1-4410-9732-ae36ae82d53c","resolution":{"observed_at":"2026-08-15T19:24:44.078241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.055223Z","title":"Multimodal Transformer for Unaligned Multimodal Language Sequences,","venue":null,"work_id":"a0ed0c20-95a8-4bb9-ae1f-fe4fe1964da9","year":2019},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.522791Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:974e28d0c619940db72f7d8fca1a1a962ca96abcb6fe10f9e5b238ecd930b6f1","observation_id":"32efbc5e-5cac-43de-afb7-d41906ca3844","resolution":{"observed_at":"2026-08-15T19:24:44.060827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.037199Z","title":"MM-TTS: Multi-Modal Prompt Based Style Transfer for Expressive Text-to-Speech Synthesis,","venue":null,"work_id":"525faaa1-4c61-4a1b-9490-fdda5a3f1c1e","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.527689Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:16a36f1b442cddedd9bd38ade8187b3c93dacd1bd5941fe11f7be9f47857a1ba","observation_id":"ee20c200-d7fe-4b0d-8b73-2a133dc09926","resolution":{"observed_at":"2026-08-15T19:24:44.042914Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.019006Z","title":"Face2Speech: Towards Multi-Speaker Text-to-Speech Synthesis Using an Embedding Vector Predicted from a Face Image","venue":null,"work_id":"68c6b4e6-8854-499e-a17f-c8fd1fcce55b","year":2020},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.532663Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:3c68c13b56ef74965fae214cf921267d159d511dee18b9302e5f355b691bb9e0","observation_id":"6a3d15c0-2bb8-4b13-8a99-002a745e76f8","resolution":{"observed_at":"2026-08-15T19:24:44.024907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:44.000488Z","title":"Imaginary V oice: Face-Styled Diffusion Model for Text-to-Speech,","venue":null,"work_id":"93df65f6-f66e-4bd6-8474-eac312eccf04","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.538046Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:da51374ad03a1e3cf9bd9d30538764e1b450a393832651fbd5854c0c8c7b801e","observation_id":"737b1011-0653-4d9a-95b6-266124a944b8","resolution":{"observed_at":"2026-08-15T19:24:44.006035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.982493Z","title":"Face-based V oice Conversion: Learning the V oice behind a Face,","venue":null,"work_id":"8a7b517a-634d-48ec-bf2d-2b6091b8cd2b","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.542951Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:ca123af1a59a071022f7f49b1f47817e4d7f98e9ddb20c27ba1ad32619e93ace","observation_id":"69cb87c1-020e-4e9a-a8af-7ddcc8a71166","resolution":{"observed_at":"2026-08-15T19:24:43.987917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.964958Z","title":"EALD-MLLM: Emotion Analysis in Long-sequential and De-identity videos with Multi-modal Large Language Model,","venue":null,"work_id":"95c32e5d-00fc-4b2c-85a0-2585dd9c6d3f","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.547738Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:12e1bbb82cc8150c0875f58a06c2c2f3e1131c01f5a668ba0c431c16a2788833","observation_id":"8e8ae700-4c78-4e78-9ee8-315aa8d2c3bf","resolution":{"observed_at":"2026-08-15T19:24:43.970554Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.947370Z","title":"Prompt-to-Prompt Image Editing with Cross Attention Control,","venue":null,"work_id":"19354d6e-35f1-452f-a74a-aabf3ae0f8f3","year":2022},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.552971Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:65bc5e532342767e8924a76bb47220dc4560be11fcd6a1a895d4b66f1c4899d3","observation_id":"9d4793a2-8c95-464a-9375-920ad525d446","resolution":{"observed_at":"2026-08-15T19:24:43.952962Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.929839Z","title":"On the Opportunities and Risks of Foundation Models,","venue":null,"work_id":"a70ca6b1-5523-4e55-9801-e3e28fee3167","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.557812Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:2fc008761f70472fbc23d08a605044a71f7b49d46d5822e51abfe4ac62e0a4b1","observation_id":"126278fd-f50f-43c1-b1eb-eb724913a4e3","resolution":{"observed_at":"2026-08-15T19:24:43.934970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.562721Z","title":"Learning Transferable Visual Models From Natural Language Super- vision,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.562721Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:23f73aa05ea2209971bdd32f8483a27f6aa4e50eb091620bfcc5344143f92ea0","observation_id":"31ef3ae8-e367-4b71-b60a-6c4ade301bbb","resolution":{"observed_at":"2026-08-15T19:24:43.562721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.899087Z","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision,","venue":null,"work_id":"9d4e857b-1009-432d-bdcd-2af55ae1559c","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.567538Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:3a607219668680680fee3f1fbe9ded1a9072a5f3f77fe604191612f462c95145","observation_id":"3d06d6dc-79dc-4c42-9330-c2ae289fea9f","resolution":{"observed_at":"2026-08-15T19:24:43.904770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.881551Z","title":"Video (language) modeling: a baseline for generative models of natural videos,","venue":null,"work_id":"e6b4e0ad-73bf-49f3-b9fb-7446d5e657bb","year":2014},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.572588Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:7a2f4058699003911641fc5810a1878908499fdde0f2d20ed2d500c817862d9c","observation_id":"5c3ab2d0-8748-4e9b-bcc7-e1f0c1626e44","resolution":{"observed_at":"2026-08-15T19:24:43.886972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.863617Z","title":"ESCoT: Towards Interpretable Emotional Support Dialogue Systems,","venue":null,"work_id":"785d55d1-1d59-4c64-a7c1-f926020f1efb","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.577944Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:f768c8f470633729160f2ce4f3f43e2b353a53434e5653f9684660d687070dce","observation_id":"e1701d47-786d-4fa4-a8da-2256797fce40","resolution":{"observed_at":"2026-08-15T19:24:43.869201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.845700Z","title":"Chain-of-thought prompting elicits reasoning in large language models,","venue":null,"work_id":"34b4307e-9ec3-4500-84f0-9dc10450b161","year":2022},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.582760Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:9ae77eb205a9943021e0731fc586fb0f3ad8345521259fcc47c6933600fade86","observation_id":"7c76fb4c-2b38-4fa2-bf05-0294ed236ded","resolution":{"observed_at":"2026-08-15T19:24:43.851199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.828087Z","title":"AutoFoley: Artificial Synthesis of Syn- chronized Sound Tracks for Silent Videos With Deep Learning,","venue":null,"work_id":"a038bc92-73f0-4168-b8b6-79a721bb2ca6","year":1907},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.588104Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:7392df11dd0e13df62ead4281e6f0266cf3fad6903f9fd64d51ad381cb5d275d","observation_id":"497cc129-74de-4c4c-83d6-34f02c58d374","resolution":{"observed_at":"2026-08-15T19:24:43.833608Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.810816Z","title":"MM-Diffusion: Learning Multi-Modal Diffusion Models for Joint Audio and Video Generation,","venue":null,"work_id":"4996bb6e-565d-4263-a7f2-7855e0aefed3","year":2022},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.593219Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:a12c566e093c0fd1bedb4e0f36bd0bbec56b414fcdb2beeb9732ff8d656561f7","observation_id":"9ba9a32c-6ca3-47eb-8882-12b75e67327d","resolution":{"observed_at":"2026-08-15T19:24:43.816183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.793710Z","title":"V ocoder-Based Speech Synthesis from Silent Videos,","venue":null,"work_id":"66ef660a-b24a-4780-b5e1-294d30333792","year":2020},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.598464Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:6cdfe6e1136275fdb784d7c267bb90c1fcec4b350713d422e3ea67fdc9a220ae","observation_id":"594c3961-62c0-4484-bab1-3a1e523497aa","resolution":{"observed_at":"2026-08-15T19:24:43.799106Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.774564Z","title":"Sonicvisionlm: Playing sound with vision language models,","venue":null,"work_id":"f51d0b8a-8537-41e8-8ce2-5f8a3c436e8f","year":2024},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.603410Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:7191336b9517a92132c60c9903c811c317948c19996d93b5213b28ad46461860","observation_id":"24afd26e-ed5a-4f7d-8684-ae19e4c93433","resolution":{"observed_at":"2026-08-15T19:24:43.780714Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.755453Z","title":"‘The problem-centred expert interview’. Combining qual- itative interviewing approaches for investigating implicit expert knowl- edge,","venue":null,"work_id":"8b01fe12-699b-40c8-9738-eff0a973721d","year":2021},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.608825Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:383c0333d2334c5df7fd7e3ac14ea99b54bccd00cd8641c160f3da2e748a77a6","observation_id":"7b40f8ce-19fe-4f2d-a40a-3538288d9573","resolution":{"observed_at":"2026-08-15T19:24:43.761256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.737893Z","title":"Pleasure-arousal-dominance: A general framework for describing and measuring individual differences in Temperament,","venue":null,"work_id":"6d0f0923-17f1-4eb2-acea-8f3c9947c491","year":1996},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.614045Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:fbebc9b6b467478cff6a7d83880112f38399ccdc86db736a8ab9335d3ade1a0c","observation_id":"d5c21d8a-1c4c-4770-8be6-0cf24d659f51","resolution":{"observed_at":"2026-08-15T19:24:43.743018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.720590Z","title":"Pleasure, Arousal, Dominance: Mehrabian and Russell revisited,","venue":null,"work_id":"1f56f7e7-6878-4e63-9a84-32ccd1ad202f","year":2014},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.619728Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:b0073bba81c623872736a469536111ddca0bb75e696d1ba4322c6a67e7d33c21","observation_id":"2a71155e-bb1e-4264-a9c3-f76e58c61d21","resolution":{"observed_at":"2026-08-15T19:24:43.726248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.702457Z","title":"Gemini: A Family of Highly Capable Multimodal Models,","venue":null,"work_id":"8a5066c2-fa07-45e2-82db-2873d7f82337","year":2023},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.624968Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:73b8098dd6dbdc55a311c933c6f20226661973e5d27280b86194ecd0a08f1ee1","observation_id":"75de29f2-1089-402a-a2d7-e7f90cc8894c","resolution":{"observed_at":"2026-08-15T19:24:43.708286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T19:24:43.683492Z","title":"Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks,","venue":null,"work_id":"13eedeeb-ed4b-4cff-8b4b-05008104a3b6","year":2019},"citing_paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T19:24:43.629693Z"},"links":{"citing_paper":"/paper/2506.16716"},"observation_digest":"sha256:4167fffe5c2a64f3025f92d255288b5e35fa6159251afd1638de13790f3cddbc","observation_id":"1b1aa4ce-d4d3-4b2c-8712-18e69e1f7595","resolution":{"observed_at":"2026-08-15T19:24:43.690513Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.16716","last_updated":"2025-06-20T03:24:34Z","latest_version":1,"primary_category":"cs.HC","snapshot_observed_at":"2026-08-15T19:17:47.662491Z","submitted_at":"2025-06-20T03:24:34Z","title":"V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos"},"reference_resolution":{"displayed":47,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":3,"verified_exact":0,"verified_fuzzy":44},"total_outbound_references":47},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 47 of 47 outbound references and 0 inbound Pith citation observations for arXiv:2506.16716."}