{"as_of":"2026-08-13T12:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e802fae67bf018fab44ed8fd49bb0ba8795f6d82b670997c5c8faef538a3a197","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:40:11.848020Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:39:46.886300Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T11:40:12.056485Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"cited_work":{"arxiv_id":"2506.02088","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.02088","snapshot_observed_at":"2026-08-07T11:40:12.056485Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","venue":"cs.SD","work_id":"8001bf5c-13fe-4d95-890b-cb21d83bd18c","year":2025},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.886300Z"},"links":{"cited_paper":"/paper/2506.02088","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:c093cda484cf71b6c0b0ae798abec31a19eac027cd58678cb7185079e74eb75b","observation_id":"b0fd2927-ab92-4ea7-b6d2-df576f3a353d","resolution":{"observed_at":"2026-08-07T11:40:12.074820Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02088/citation-record","integrity":"/paper/2506.02088/integrity","json":"/paper/2506.02088/citation-record.json","paper":"/paper/2506.02088"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.813882Z","title":"Early SER relied on hand-crafted features but struggled with real- world generalization [2]","venue":null,"work_id":"28a486ba-2220-47bd-8f91-a9a84090d304","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.828246Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:e1ddc8fad7d8798ee3d0676413e834095b7e9e5b6005bba75a6df9b309b448a2","observation_id":"cbc91fd8-2696-431c-83e7-60c230cb87b5","resolution":{"observed_at":"2026-08-07T11:40:58.900210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"cited_work":{"arxiv_id":"2506.02088","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.02088","snapshot_observed_at":"2026-08-07T11:40:12.056485Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","venue":"cs.SD","work_id":"8001bf5c-13fe-4d95-890b-cb21d83bd18c","year":2025},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.886300Z"},"links":{"cited_paper":"/paper/2506.02088","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:c093cda484cf71b6c0b0ae798abec31a19eac027cd58678cb7185079e74eb75b","observation_id":"b0fd2927-ab92-4ea7-b6d2-df576f3a353d","resolution":{"observed_at":"2026-08-07T11:40:12.074820Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.659330Z","title":"The hidden states of the last layerLof the text encoder are denoted byZ L T (j)for positions j= 1,","venue":null,"work_id":"91e78c0a-8895-4a59-93ca-a567c0d06b78","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.937316Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:6eb215e9f7c06fd9ed5bf5d62db8263a6fc9058c9f16386b34d4178f96287814","observation_id":"d0e4715b-d7fe-41df-a432-fdb412559c61","resolution":{"observed_at":"2026-08-07T11:40:58.725410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.626319Z","title":null,"venue":null,"work_id":"a62edddf-e725-4da4-a364-7ec92f78a632","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.962061Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:f5739021a03002ba0c0db4c510c03afd856a8b48a8cf04645581071b01eeda9e","observation_id":"568c6db1-c9bf-445c-b9e1-92b773217c62","resolution":{"observed_at":"2026-08-07T11:40:58.649776Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.443731Z","title":"We report results for unimodal speech models, bi- modal fusion with text, prosodic and spectral feature integra- tion","venue":null,"work_id":"56593075-ba1b-4b70-8084-8883969b0890","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.019513Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:90cf38871bbfadfb99248accdf2a925688621fa3e28b74f79bb10f7437641029","observation_id":"33001ac9-da6f-4328-9c06-f8c518f3b824","resolution":{"observed_at":"2026-08-07T11:40:58.509032Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.292421Z","title":"Our evaluation of unimodal models demonstrated the strong performance of Whisper and XEUS, highlighting their robustness for SER in spontaneous speech","venue":null,"work_id":"d47b58a4-ef1b-4a75-bccc-1eb08c1a166e","year":null},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.249227Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:1f68dd5419b6757da603cd4ef9d50687fd63b8c9d7edb8ae5ba4816478ceb2f4","observation_id":"a15c6334-fba5-4b40-977e-75b1be96e4a9","resolution":{"observed_at":"2026-08-07T11:40:58.356842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.153649Z","title":"We also thank the Artificial Intelligence Lab at Re- cod.ai, the Institute of Computing, University of Campinas","venue":null,"work_id":"022effbf-6002-4c36-a858-e2fe68378742","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.380433Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:d7b9553838e2c8acb38ccae88f8f185abbe0dd738133978298de2131ad5285c9","observation_id":"8a356d0a-2d75-4065-b968-ab684529c6e7","resolution":{"observed_at":"2026-08-07T11:40:58.217464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:58.085467Z","title":"Affective computing mit press,","venue":null,"work_id":"e73d1712-5573-424a-a829-9cbb864bca08","year":1997},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.456692Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:4be9647858c0a0a42813098c56bc897208d6293f3eb277861e722a1db77004a6","observation_id":"bd61b554-8a33-449d-b640-1d6fbccfcf00","resolution":{"observed_at":"2026-08-07T11:40:58.093220Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:47.566911Z","title":"Iemocap: Interactive emotional dyadic motion capture database,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:47.566911Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:067dc60f1cc2f92941786ab3b98b5fca16fcc41332006a64df3ffbb709b90efd","observation_id":"95edbc09-aad6-4d1d-ae6e-d62d26767b08","resolution":{"observed_at":"2026-08-07T11:39:47.566911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.973677Z","title":"Every rating matters: Joint learning of subjective labels and individual annotators for speech emotion classification,","venue":null,"work_id":"bb3c803f-7f38-42bc-9c81-4c60c6bdaf7b","year":2019},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:48.879938Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:052231cb8e43f481f3ce7ab6bed2f5c2b109c346beb05256f37f9e3cab0856ca","observation_id":"575e1dbe-4310-41de-a8e4-6c02731c8573","resolution":{"observed_at":"2026-08-07T11:40:58.052371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:51.667321Z","title":"Speech emotion recognition using self-supervised features,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.667321Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:0599377429eb0e0a4ffd9fc907d910ec9b03658f58ae587abf64b8e5f130b797","observation_id":"5a988f1c-8399-4005-9333-b47543d5203f","resolution":{"observed_at":"2026-08-07T11:39:51.667321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.900272Z","title":"Chakraborty, M","venue":null,"work_id":"fccf5fc0-a0f7-4055-b364-f80fe57a64a2","year":2017},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.789972Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:2d632d94a41f690adbbc24a423362a790fbf1b6b25bee2ad491dc45dd9ff6b7d","observation_id":"4aa89ce1-a259-4b5d-a722-c3b2640b180e","resolution":{"observed_at":"2026-08-07T11:40:57.908096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.796562Z","title":"Speech emotion recognition with multi-task learning,","venue":null,"work_id":"8cfcf4dc-7410-47b6-abd8-72d3b4ec4e71","year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.886757Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:c5a6bd7c23dee804de9c1aae4f8364da09674fa79bda6199121a185b29077a91","observation_id":"f8ca3961-213f-4b2b-91ec-7514c4909d14","resolution":{"observed_at":"2026-08-07T11:40:57.829289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:57.708229Z","title":"Improving speech emotion recogni- tion using self-supervised learning with domain-specific audiovi- sual tasks,","venue":null,"work_id":"66c02912-f665-4b9a-8ab4-3c541e5bd203","year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:51.999237Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:51496be79b61ee08d14070679f9ff6b1723ab52da357266684cf2ea5bb609036","observation_id":"af2b8935-da7c-4af9-ab15-bbac4f956498","resolution":{"observed_at":"2026-08-07T11:40:57.755073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:56.052863Z","title":"Odyssey 2024-speech emotion recognition challenge: Dataset, baseline framework, and results,","venue":null,"work_id":"3431ade5-553a-4f50-a76c-faec51528625","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.100807Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:c5a208211d6af947ffcf85908eac39f8e963e8f1f0a59959074fae8fa04f65ff","observation_id":"dc666eda-a686-4562-ac18-c5b4e6d3a2df","resolution":{"observed_at":"2026-08-07T11:40:57.518830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.179588Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.179588Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:cec95be3d9035b9396230cfb99dbdb00e408fbe3d0a84ae045a6b23d93bc65c9","observation_id":"e4a34ba8-918f-4d92-a487-dfd354d892a4","resolution":{"observed_at":"2026-08-07T11:39:52.179588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.259720Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.259720Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:4e08e7f0a9f987a5f21b6da9e22d21b9e98c22c2274584aa2b1de6ab8fef7cf3","observation_id":"cafa7854-99b7-47db-b70b-92195a3418cd","resolution":{"observed_at":"2026-08-07T11:39:52.259720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.329379Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.329379Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:cd7e9638c389e3763f25311f7b0c39fa042f047c32e56633c8fc1f7068e9842c","observation_id":"82513e37-8794-4a55-986b-5d009f0a68c3","resolution":{"observed_at":"2026-08-07T11:39:52.329379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.399609Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.399609Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:d2ace477d1df446cd3d411869eb411de8c135c8c46273f1433361b656fc926df","observation_id":"814768dd-203d-4df1-bc62-9d272e49614e","resolution":{"observed_at":"2026-08-07T11:39:52.399609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:39:52.465169Z","title":"Towards robust speech representation learning for thousands of languages,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.465169Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:fe2f290ba9ece7750580b981b267ccbcb504c92c5640f9b6d45de637ab048b72","observation_id":"609fc84d-5d93-4fe4-a39b-a0eb1dd6b082","resolution":{"observed_at":"2026-08-07T11:39:52.465169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.920125Z","title":"A robustly optimized BERT pre-training approach with post-training,","venue":null,"work_id":"5cc7a949-fb7e-4073-a3b5-903203b7bbe9","year":2021},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.554243Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:eed1c49a0ad758722abbc255d9f9b545fb69799dcf9d5c8603189ae9ba8cb475","observation_id":"fb018c13-febc-43f5-9629-7294bac6d95a","resolution":{"observed_at":"2026-08-07T11:40:55.952387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.01954","last_updated":"2023-05-03T08:11:25Z","snapshot_observed_at":"2026-08-13T11:50:16.718312Z","submitted_at":"2023-05-03T08:11:25Z","title":"SeqAug: Sequential Feature Resampling as a modality agnostic augmentation method","version":1},"cited_work":{"arxiv_id":"2305.01954","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.01954","snapshot_observed_at":"2026-08-07T11:40:12.003559Z","title":"SeqAug: Sequential Feature Resampling as a modality agnostic augmentation method","venue":"cs.CL","work_id":"6e99f6d9-251f-40ee-8455-41fd9343b685","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:52.777099Z"},"links":{"cited_paper":"/paper/2305.01954","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:a652ef949e5f7e5cc2650bea9424b5c9d0007cb605cd1411990331f82110bfd0","observation_id":"aa5f7fc5-f77c-4c38-8d0c-800414f835c5","resolution":{"observed_at":"2026-08-07T11:40:12.025604Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:53.234553Z","title":"Enhancing cross-language multimodal emotion recognition with dual attention transform- ers,","venue":null,"work_id":"712c8579-29d9-4dd5-8b79-56c21511d91e","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.019255Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:6b7d341de14f117d80d1f552c9f40e2561688290469865d178ddb525da6efdac","observation_id":"6c227393-77bb-4ef8-ad33-c932444ebe03","resolution":{"observed_at":"2026-08-07T11:40:54.883253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:36.106956Z","title":"Ced: Con- sistent ensemble distillation for audio tagging,","venue":null,"work_id":"26ac8abe-9879-4458-9a69-665c3a5bd411","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.103227Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:fec292f52eac03ed6cdaabdbf4e67fddfdac2ec9cdd7f958188d4e1ddf6f3274","observation_id":"57f0aca7-3596-43b6-8b87-bf654988b716","resolution":{"observed_at":"2026-08-07T11:40:46.567879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.06910","last_updated":"2024-01-09T11:45:34Z","snapshot_observed_at":"2026-08-13T12:03:09.630883Z","submitted_at":"2023-04-14T03:25:00Z","title":"HCAM -- Hierarchical Cross Attention Model for Multi-modal Emotion Recognition","version":2},"cited_work":{"arxiv_id":"2304.06910","doi":null,"metadata_source":"pith","pith_arxiv_id":"2304.06910","snapshot_observed_at":"2026-08-07T11:40:11.955002Z","title":"HCAM -- Hierarchical Cross Attention Model for Multi-modal Emotion Recognition","venue":"eess.AS","work_id":"d66b6061-c7da-48c1-83f1-a558fa78b36a","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.213116Z"},"links":{"cited_paper":"/paper/2304.06910","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:bfbec0dc32fea6d86dcc745e6dda4f4091b4796e0469c2352000260d7dfd7997","observation_id":"cfa2a6ae-afd7-4b88-a60d-6e728af99904","resolution":{"observed_at":"2026-08-07T11:40:11.978878Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.993290Z","title":"Graph attention networks,","venue":null,"work_id":"81664a14-7051-4159-9066-495a895e2bcc","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:53.374984Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:b695e8d0d890d0004eeb9b59206e776444e87ca8aa0c70cb7645e944c9845e23","observation_id":"adf8d24e-0b38-43d3-8eb2-5c3958ba2a54","resolution":{"observed_at":"2026-08-07T11:40:36.044619Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-08-11T06:21:56.129166Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-07T11:40:07.102687Z","title":"Glu variants improve transformer,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:07.102687Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:ed863c0b65c5243cdb9fc99ed5b6cbafa0cbed1f6329f16587506502fdd0d324","observation_id":"73217c7a-5c82-49d8-8a29-7fffc3b96f74","resolution":{"observed_at":"2026-08-07T11:40:07.102687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:35.812409Z","title":"Espnet: End-to-end speech pro- cessing toolkit,","venue":null,"work_id":"2d026bc1-3e4e-46b4-8a50-c5cca55acd4d","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:08.043740Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:70d5c4362da81c66c7357b1b5525791388ff1024f589aea6756a5bb6bc7be009","observation_id":"2e2d95a2-2ebd-46ec-8686-ddc1d47f9b94","resolution":{"observed_at":"2026-08-07T11:40:35.916452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:09.312368Z","title":"Less is more: Accu- rate speech recognition & translation without web-scale data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.312368Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:9f599259839473c863a3c1a5ae6290fdb564e5ed56fe8a34effc84f4f6b919a5","observation_id":"571a2128-c921-4234-92c9-21618fd4f1e6","resolution":{"observed_at":"2026-08-07T11:40:09.312368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:55.549542Z","title":"1st place solution to odyssey emotion recognition chal- lenge task1: Tackling class imbalance problem,","venue":null,"work_id":"ff5cfa22-8e4d-4ca5-9631-3a66fd76b149","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.642774Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:ba428f3dc7047165c5debe0554d73243cb7b553ed3d036893c39fce86b5b5d9a","observation_id":"738e5e1b-7056-46d6-ab15-de24cbf70da5","resolution":{"observed_at":"2026-08-07T11:40:55.870025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:33.534336Z","title":"Fundamental frequency ex- traction in speech emotion recognition,","venue":null,"work_id":"d898edcb-437f-4db3-a406-07e453fdd95d","year":2012},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.779236Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:08122afc6b862d99e5ee86fc0f2d4248f21bcda2febb57b657b39220f6550193","observation_id":"c2ef2ada-153e-4b7b-b0f4-8a58e4ec2f0f","resolution":{"observed_at":"2026-08-07T11:40:35.694011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:30.830689Z","title":"Autoregressive neural f0 model for statistical parametric speech synthesis,","venue":null,"work_id":"506b7420-7c15-4e05-b2a8-0047189c3e46","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.869009Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:c6f011de9d78b680feb981589485222d1dc4dc9ad73abe2982def3d07b7916d8","observation_id":"b59c1520-ded0-4b57-84eb-320e3a08162b","resolution":{"observed_at":"2026-08-07T11:40:31.875345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:23.564544Z","title":"Rmvpe: A robust model for vocal pitch estimation in polyphonic music,","venue":null,"work_id":"1c9fff7c-6b3a-46bd-a442-b3ad964239f6","year":2023},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.900326Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:0cb1f554df4e66f55c6bd414c28901970083879144b90398d8b1758035b5eb07","observation_id":"f9bb7dae-4b5b-4664-94b4-2ea372393358","resolution":{"observed_at":"2026-08-07T11:40:23.794768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:21.613155Z","title":"Enhancing skin can- cer diagnosis using swin transformer with hybrid shifted window- based multi-head self-attention and swiglu-based mlp,","venue":null,"work_id":"d77c6639-6288-45ea-a471-a6d3a30936b8","year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.918778Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:00f541e16a127d7709cdaa9a03c5b78160cd3742ea5ec03516ddbbd948a81def","observation_id":"b5824415-23d1-4220-ac0e-c03afc0ea926","resolution":{"observed_at":"2026-08-07T11:40:23.219875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.214296Z","title":"Searching for activation functions,","venue":null,"work_id":"8782f5ca-43ca-412e-8c59-ff108c60ce46","year":2018},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:09.980654Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:7ecbf4b5678c1e8e4f6ca2ba5d8331b1cc29afdbad97fa004766ea0aace387d1","observation_id":"61f5b1ec-e191-4293-afb4-86cceefebbc0","resolution":{"observed_at":"2026-08-07T11:40:12.252580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:11.029630Z","title":"Decoupled weight de- cay regularization,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.029630Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:733257ecda6f0633c33e232a7fa29ee2efc173d0303ef2bb9ece8f81af6cfaec","observation_id":"dec73175-a10b-4b58-8360-d52c414541d0","resolution":{"observed_at":"2026-08-07T11:40:11.029630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:11.634578Z","title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.634578Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:a7849ad5a976fe6bc789c3e71f24b404b975fce8a9b9c57ea928c49aee26d3d0","observation_id":"072fc2c8-d40f-4cfd-8b25-d6b33dbf5e32","resolution":{"observed_at":"2026-08-07T11:40:11.634578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:40:12.111418Z","title":"Focal loss for dense object detection,","venue":null,"work_id":"68afa429-e4d8-4ac0-a60a-db1e52b2a1d8","year":2017},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.768796Z"},"links":{"citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:fdb88b7eba8c12b902607fff0caedf90050a61b8a3719fa04923b3aa44da20d2","observation_id":"b66fe9de-cb41-4a86-b269-35a9afcb68dc","resolution":{"observed_at":"2026-08-07T11:40:12.140211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05672","last_updated":"2024-02-08T13:47:50Z","snapshot_observed_at":"2026-08-12T15:58:37.148545Z","submitted_at":"2024-02-08T13:47:50Z","title":"Multilingual E5 Text Embeddings: A Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05672","snapshot_observed_at":"2026-08-07T11:40:11.848020Z","title":"Multilingual e5 text embeddings: A technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:40:11.848020Z"},"links":{"cited_paper":"/paper/2402.05672","citing_paper":"/paper/2506.02088"},"observation_digest":"sha256:107d537992d333f6be24dbb1507245109e4f3f394e5afbfb3e652a040e61b034","observation_id":"75fdba96-0af3-4525-93be-145621894b8f","resolution":{"observed_at":"2026-08-07T11:40:11.848020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02088","last_updated":"2025-06-02T13:46:02Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-12T09:45:49.802138Z","submitted_at":"2025-06-02T13:46:02Z","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":13,"verified_exact":3,"verified_fuzzy":23},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 1 inbound Pith citation observation for arXiv:2506.02088."}