{"as_of":"2026-08-18T04:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:72403b36d8d9547546d45e10e3abc1b036eda1a27fc45ee168571dca8e617413","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T19:43:00.184577Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.04845/citation-record","integrity":"/paper/2507.04845/integrity","json":"/paper/2507.04845/citation-record.json","paper":"/paper/2507.04845"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"records/1555977","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.371229Z","title":"Alpha” α and “Beta","venue":null,"work_id":"cd3d3e46-706f-46bc-aeca-d5dd029a0a1e","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.031707Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:300ddc277edfbefe62cd2cb13de72ae39ca82dd2c8710b00884ec455032252a3","observation_id":"5a7653ba-d2fc-4eb8-8596-656f5a9231eb","resolution":{"observed_at":"2026-08-06T19:43:00.375509Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.745897Z","title":"Alpha” ∈ RTα×dk is combined with another modality “Beta","venue":null,"work_id":"fa262b07-b094-49d4-8dfb-73a9da407ace","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.035565Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:06e22c7cef0cdcb1b97642898ea343ef1f4b388a093c0f85b3b4d7a3965d10cb","observation_id":"10b35c16-0870-4261-9298-ce3d3db7658a","resolution":{"observed_at":"2026-08-06T19:43:00.749121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.728836Z","title":"So, we adopted the inter-channel level difference (ILD) as the primary spatial feature for the SELD encoder, alongside log mel spectrograms computed independently from each channel","venue":null,"work_id":"8ab4de83-700f-4152-bdfb-0c7c80ebafa8","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.042917Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:87e1308c4bdf75733bf2e1d536b8570bc7e4e9de6bdc880e944efd9aa6f46a10","observation_id":"83d0c384-8d0e-4329-8e08-898ae9d40193","resolution":{"observed_at":"2026-08-06T19:43:00.731958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.720315Z","title":"Knock” and “Bell","venue":null,"work_id":"a531a770-9cf5-4da0-8a20-f68a619c0676","year":null},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.046019Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:4537ad550d60aeb6701e81843f5ded292dfc8f0a5d8219f1cb584e3900562cdd","observation_id":"65dc058f-e69c-4d68-b656-20342f104d57","resolution":{"observed_at":"2026-08-06T19:43:00.723311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.711290Z","title":"off-screen","venue":null,"work_id":"69b454df-6127-4c3d-a75f-f91de6ec9e75","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.049821Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:110f430a30a0063b70cd45f186bbc00b23d35797dbe677de33d822cb426eb298","observation_id":"d1e8fb21-11e9-487f-8017-b3b7365235b2","resolution":{"observed_at":"2026-08-06T19:43:00.714267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.702275Z","title":"Our approach leverages a model that integrates semantically rich feature embeddings from CLAP and OWL-ViT, fused through an adapted Conformer architecture","venue":null,"work_id":"2000a062-f978-4224-971c-bfaa62cd75de","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.053495Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:b03f9f3f98d9db3e6e6ffb938a884a053649a11e23ca0899624122350e555018","observation_id":"1b64737d-b1f6-4927-bdcf-4af7c361af45","resolution":{"observed_at":"2026-08-06T19:43:00.705504Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.638530Z","title":"A dataset of dynamic reverberant sound scenes with directional interferers for sound event localization and detection,","venue":null,"work_id":"7314aad4-26a0-4f37-aa2b-4a1a0b8bd943","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.074924Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:c732b9fac64c49fa0bcba144a196f577ac48ce8bb024ab3e4cc7209d5b235169","observation_id":"fc6b3b77-f741-4906-bf21-af82a4b5a976","resolution":{"observed_at":"2026-08-06T19:43:00.641911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.693832Z","title":"Sound event localization and detection of over- lapping sources using convolutional recurrent neural networks,","venue":null,"work_id":"14af17d5-5d3c-4f17-b48f-74c7f827eb3b","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.056268Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2abb48b2c083733c71c4a52b9db0f1f96e1de844afeb680b16a3c5f78185f725","observation_id":"f2bb4d1e-ba8c-43a4-ba61-f77cc13ec2c5","resolution":{"observed_at":"2026-08-06T19:43:00.697013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.684112Z","title":"Sound event detection using spatial features and convolutional recurrent neural network,","venue":null,"work_id":"7cc51aed-4d36-4592-951f-76f99b9a069d","year":2017},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.059111Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:80bd76c0b0d5222c428fd79a5438d07f40575f4c55ce8c96db829f000e55ad38","observation_id":"c0909112-3edc-4da9-bcd7-e4b10786164b","resolution":{"observed_at":"2026-08-06T19:43:00.688103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.675512Z","title":"Direction of arrival estima- tion for multiple sound sources using convolutional recurrent neural network,","venue":null,"work_id":"b56bdf5a-52e7-4144-8414-caf9429b7937","year":2018},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.062926Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:eb046653d16caa7a90d05303eaab8e96f935506e1951f8f1d60ebc664e273d4b","observation_id":"ac1f2b21-fe6d-42a7-9ee4-26507c174c94","resolution":{"observed_at":"2026-08-06T19:43:00.678456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.665927Z","title":"Audio-visual cross-attention network for robotic speaker tracking,","venue":null,"work_id":"6090f467-a248-4992-98e5-2870e040de4a","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.065774Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:87582490d29e1acc04bb64d92bb49adbfdc9a3f8a5cdec4893e179ed97ba5707","observation_id":"c34b1ae6-2317-4aad-920f-addef3746d82","resolution":{"observed_at":"2026-08-06T19:43:00.669262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.657288Z","title":"ForecasterFlexOBM: A multi-view audio-visual dataset for flexible object-based media production,","venue":null,"work_id":"f9f0ec25-649d-405a-a301-f994ed233a9f","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.068451Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:3c30186b6fec6489016a4ee58de389ca9d193f67efe99cdcde0bf425fc65da8a","observation_id":"8d9b70fb-5c78-4a22-9a82-08d1321ac6bb","resolution":{"observed_at":"2026-08-06T19:43:00.660887Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.648202Z","title":"A dataset of reverberant spatial sound scenes with moving sources for sound event localization and detection,","venue":null,"work_id":"943c6a96-9aba-4b9a-bd08-770732782bfd","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.072104Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:762ca27aa0319e6c0eb6374b1c35ef36d307f9b4b0e66ae0c27d31f651a75e4b","observation_id":"56cd3fcb-8c7b-468f-b956-a8dcf8bc5cd7","resolution":{"observed_at":"2026-08-06T19:43:00.651703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.572716Z","title":"Simple open-vocabulary object detection,","venue":null,"work_id":"ecbeb797-1590-43ae-a95f-581104008acb","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.094582Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:11673f0741b2b4fcaa272ec81f238a5fe06d0988f2fc70ebbdc41e85b786ac26","observation_id":"f2f879f9-5997-4a47-9940-6c70b2015ae4","resolution":{"observed_at":"2026-08-06T19:43:00.576894Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.629643Z","title":"STARSS23: An audio-visual dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events,","venue":null,"work_id":"0381b012-53f0-4131-b659-67a505177546","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.077587Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:92a8c4729a25937fbaee713987327d3509f5a103b1264fe57552b2fe5175d363","observation_id":"aa2dd003-c222-4785-93bc-71f39380c1c0","resolution":{"observed_at":"2026-08-06T19:43:00.632898Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.620001Z","title":"Baseline models and evaluation of sound event localization and detection with distance estimation in DCASE 2024 Challenge,","venue":null,"work_id":"64db25be-477a-4d9f-9199-8d401e24c5c8","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.080652Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d96cafcaa91a67ed6b3586d2cf53731242f8f6c95c356ae9e7cee6ae2ead9e87","observation_id":"28c6262a-6740-43a5-9acb-ad5100ff0ccf","resolution":{"observed_at":"2026-08-06T19:43:00.623965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.611280Z","title":"Language models are few-shot learners,","venue":null,"work_id":"52376973-aaa3-4909-a0f2-c8a0d28a9d15","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.083383Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:75d62c5b42f983f6687bee13384a822af5f365860eae927695f828289b4fe528","observation_id":"f2c1a551-b1ca-4cbe-9c21-42265ee38c68","resolution":{"observed_at":"2026-08-06T19:43:00.614165Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.602444Z","title":"Learning transferable visual models from natural language supervision,","venue":null,"work_id":"2191431d-16a2-4b25-9ca2-a311efdca320","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.086063Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:16ba235776603aab900d6bd82ef1b5a3494dbda9f7959fe8a5e4ac1d89ab1a36","observation_id":"6aa9b7e1-5188-4b80-8ea1-2d1bee94342e","resolution":{"observed_at":"2026-08-06T19:43:00.605823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.593223Z","title":"AudioGPT: understanding and generating speech, music, sound, and talking head,","venue":null,"work_id":"68ae83ba-8c6b-4d87-9549-eeb24487c387","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.088722Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:eece70523825dd796599779dc28437b9e28193a8d2f22d361fa38c0ce43c7bec","observation_id":"3d9860d6-e136-4139-af3e-7bbea2365998","resolution":{"observed_at":"2026-08-06T19:43:00.596543Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.582865Z","title":"Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation,","venue":null,"work_id":"c1b9d362-eff7-4022-adbe-31cbdbc04565","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.091720Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:8abd3143ead224a51d7ad957ddbdc606e71eccbc7c5fa8df8514eb76fead26fb","observation_id":"4b415a27-cc9e-4c4e-bc95-5728b8ad12bb","resolution":{"observed_at":"2026-08-06T19:43:00.586512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.517477Z","title":"Multi-ACCDOA: Localizing and detecting over- lapping sounds from the same class with auxiliary duplicating permu- tation invariant training,","venue":null,"work_id":"f83fddc2-5142-4ae1-b666-7beea7b97abf","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.117153Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:baa94f4447891708dc3059086f4d903fbf3abcd6a729fcb15cbb35f981e14102","observation_id":"ad210db2-3e38-4bc7-a407-ae10db5a2fc7","resolution":{"observed_at":"2026-08-06T19:43:00.520694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.562320Z","title":"A four-stage data augmentation approach to resnet- conformer based acoustic modeling for sound event localization and detection,","venue":null,"work_id":"aade8191-8b56-48fd-af06-52f903980119","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.098194Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2bf7b36deead602451a7beb09d927153ecbb46145067def05a378562d69c618e","observation_id":"ba1d2016-0e97-4ac2-9070-64b12ec35126","resolution":{"observed_at":"2026-08-06T19:43:00.565723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.551298Z","title":"Fusion of audio and visual embeddings for sound event localization and detection,","venue":null,"work_id":"f93dba78-d9fc-4693-a5a8-092297b58d7e","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.101728Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ba82f7486dd3924c929ebbc613abee51f010801655eef29db55b36286fabf788","observation_id":"44c0f989-59bd-4cad-bb15-633281ad3d8f","resolution":{"observed_at":"2026-08-06T19:43:00.555539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.540760Z","title":"Resnet-conformer network using multi-scale channel attention for sound event localization and detection in real scenes,","venue":null,"work_id":"215bd78d-7ea1-419e-a223-9045e4910150","year":2023},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.104342Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:f6c0d479f4908f722db85b7e52dfdcf3a8bbe3fe907da4302d9c47b76e7a1eb4","observation_id":"0546a630-742c-4ef4-8a65-95633ff17233","resolution":{"observed_at":"2026-08-06T19:43:00.544602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.107548Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.107548Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:dc43be5d83b689e07c04c19b5634754a2ee249a8b1f50c28bc5ab421047399ae","observation_id":"f8965118-2192-4ba6-820e-feb8a0aecda4","resolution":{"observed_at":"2026-08-06T19:43:00.107548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08644","last_updated":"2026-04-20T08:35:46Z","snapshot_observed_at":"2026-08-16T08:11:32.317130Z","submitted_at":"2025-04-11T15:43:13Z","title":"Reverberation-based Features for Sound Event Localization and Detection with Distance Estimation","version":2},"cited_work":{"arxiv_id":"2504.08644","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.08644","snapshot_observed_at":"2026-08-06T19:43:00.262298Z","title":"Reverberation-based Features for Sound Event Localization and Detection with Distance Estimation","venue":"eess.AS","work_id":"e8efb224-4f68-48d0-9c27-9111082df40f","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.110875Z"},"links":{"cited_paper":"/paper/2504.08644","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:633377ed5eca89f345470192a441fc147027428b1272908af959b26b54b3612f","observation_id":"05aaa5a4-8fd6-4f33-81b2-fc7eb91b981b","resolution":{"observed_at":"2026-08-06T19:43:00.265610Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.737072Z","title":"Yet, we believe that this alternative sacrifices seman- tic richness, as pooling across channels degrades the learned feature representations","venue":null,"work_id":"31e31cad-041b-4eb1-8374-89a9321e8311","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.039632Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ccb0cc73fa6dce6f19c5545fe61acce3442919b4d8125c775a808ee2cd2725d3","observation_id":"06001994-f574-4dd5-9e89-f73a46062fa0","resolution":{"observed_at":"2026-08-06T19:43:00.740135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.526086Z","title":"Sound event detection and localization with dis- tance estimation,","venue":null,"work_id":"6cb1d880-2328-496b-aa7c-226b31db3979","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.114566Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d2cd5526cc4cc3f5ae8200c23a5870834101a925c3b6a680452e72bb658818aa","observation_id":"33b6d5d8-0869-43a1-a9cc-f1a2cee8de1a","resolution":{"observed_at":"2026-08-06T19:43:00.529132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.508456Z","title":"Event-independent network for polyphonic sound event localization and detection,","venue":null,"work_id":"1d0878d2-f50d-424b-8f85-edee40a96c2a","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.120263Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:9920d79d33230a8edb3bd7781884b3f0a5ae16e326ac336d2c7ad74e22a51798","observation_id":"8bfbcf0a-488e-48cf-9815-40346cda19bc","resolution":{"observed_at":"2026-08-06T19:43:00.512019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.499519Z","title":"An improved event-independent network for polyphonic sound event localization and detection,","venue":null,"work_id":"5522c3e3-e70e-4626-8c08-eded0f2d8b75","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.123514Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:093d072577f75f7a8a5f0e3c9b78db27683008fec1507b5890a9c0c7aeef33c9","observation_id":"3167a9db-4b10-49c9-992c-2ca0c97da4c3","resolution":{"observed_at":"2026-08-06T19:43:00.503001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.489778Z","title":"Batch normalization: Accelerating deep net- work training by reducing internal covariate shift,","venue":null,"work_id":"9486f544-0b67-48ab-84d5-ed94cb5492db","year":2015},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.126484Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ac706db057eef55171df259359c8ce6d07bd4f59809c805252a5ffc7c13b0bcf","observation_id":"af64aa6c-50bd-4441-b49a-eaf426bcfe0f","resolution":{"observed_at":"2026-08-06T19:43:00.493043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.480474Z","title":"Leveraging reverberation and visual depth cues for sound event localization and detection with distance estimation,","venue":null,"work_id":"da917ce3-280f-4756-be36-d957bddc4f79","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.129132Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:cfe74fb17caf42fb318fab7fa674dd770d90ad7a9684c18352fd39350020fd81","observation_id":"6a0054de-a54b-4033-bbda-e337668a458a","resolution":{"observed_at":"2026-08-06T19:43:00.484523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.470944Z","title":"Deep residual learning for image recognition,","venue":null,"work_id":"33dafe3a-5c40-4f68-bf60-41c0c6401f5d","year":2016},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.132240Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:8e98046be511c5f4f2494f02d4666c880548a8bd213f0e9f0444dfb1d504f91b","observation_id":"f4c2311c-66b3-4fe3-92e3-c3aa62f8adf1","resolution":{"observed_at":"2026-08-06T19:43:00.475259Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14153","last_updated":"2024-11-21T14:18:10Z","snapshot_observed_at":"2026-08-14T03:12:01.842787Z","submitted_at":"2024-11-21T14:18:10Z","title":"MVANet: Multi-Stage Video Attention Network for Sound Event Localization and Detection with Source Distance Estimation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14153","snapshot_observed_at":"2026-08-06T19:43:00.135285Z","title":"MV ANet: Multi-stage video attention network for sound event localization and detection with source distance estima- tion,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.135285Z"},"links":{"cited_paper":"/paper/2411.14153","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d2ab6eefdf39c50218543ac64687da21bf5a96fc593036d0d8aa1c00b6345b32","observation_id":"8c269f59-3f37-4139-85da-838595967f76","resolution":{"observed_at":"2026-08-06T19:43:00.135285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.462439Z","title":"Spatial Scaper: A library to simulate and aug- ment soundscapes for sound event localization and detection in realis- tic rooms,","venue":null,"work_id":"e56cfb1d-8e27-4c56-b188-9cffbbc6dc28","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.139318Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:08b4dfd52a8ad54412439b8ccfe636397781c6bb3399abac06d083e836dc0e04","observation_id":"04cc7c39-2166-4e55-a44f-8947b7003981","resolution":{"observed_at":"2026-08-06T19:43:00.465931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.453713Z","title":"FSD50K: an open dataset of human-labeled sound events,","venue":null,"work_id":"38e61d2b-ddd7-40bc-af30-6a3e04eb3422","year":2022},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.142517Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:592fab339897a354be1b41fb47c6fa25aa776c688c08b486ebb2c9f6fb5ebe4c","observation_id":"20406617-8f72-4335-b914-07438f7dc466","resolution":{"observed_at":"2026-08-06T19:43:00.456725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.5281/zenodo.2635758","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"METU SPARG Eigenmike em32 Acoustic Impulse Response Dataset v0.1.0,","venue":"Figshare","work_id":"6e37c6c5-6dcc-4ee7-9ea9-cbe4420cb5d4","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.145300Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2b28806eea4bf539ca21337a0e36f7eb475efc49230bdcfa00850453aa0e204c","observation_id":"18230b25-9ce1-483f-9076-235d2659a590","resolution":{"observed_at":"2026-08-06T19:43:00.209421Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11882","last_updated":"2021-11-23T13:51:17Z","snapshot_observed_at":"2026-08-16T17:39:55.666543Z","submitted_at":"2021-11-23T13:51:17Z","title":"Dataset of Spatial Room Impulse Responses in a Variable Acoustics Room for Six Degrees-of-Freedom Rendering and Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.11882","snapshot_observed_at":"2026-08-06T19:43:00.148335Z","title":"Dataset of spatial room impulse responses in a variable acoustics room for six degrees-of- freedom rendering and analysis,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.148335Z"},"links":{"cited_paper":"/paper/2111.11882","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:988317f3a753493c301dfd0dec087c46327469fcd29ff658966a4d4e8281a10d","observation_id":"cd0c6474-7890-4180-ba42-9a757fcc32b8","resolution":{"observed_at":"2026-08-06T19:43:00.148335Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1612.01840","last_updated":"2017-09-05T18:38:33Z","snapshot_observed_at":"2026-08-14T21:26:53.042056Z","submitted_at":"2016-12-06T14:58:59Z","title":"FMA: A Dataset For Music Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1612.01840","snapshot_observed_at":"2026-08-06T19:43:00.151931Z","title":"FMA: A dataset for music analysis,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.151931Z"},"links":{"cited_paper":"/paper/1612.01840","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:ab25bb723fb4747c3bb13f00097b587b2fef843da5e6f879c6b37ec47a5ec24e","observation_id":"3209eada-4973-45a4-aeda-701aaf59990a","resolution":{"observed_at":"2026-08-06T19:43:00.151931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.444115Z","title":"A dataset of higher-order ambisonic room impulse responses and 3d models measured in a room with varying furniture,","venue":null,"work_id":"232650bf-bf46-40fc-927b-83560c9489da","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.155481Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:3c7a723b856375843beed134a8cd01121ebaf1524d1fe643ac9957b8117c45cd","observation_id":"4ebfa138-bec0-4575-8c44-5dd3680f18d4","resolution":{"observed_at":"2026-08-06T19:43:00.447759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.435812Z","title":"Room impulse re- sponse dataset of a recording studio with variable wall paneling mea- sured using a 32-channel spherical microphone array and a B-format microphone array,","venue":null,"work_id":"18011ac6-49c5-47f7-9363-3e7196a1c0aa","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.158221Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:64fc1d2e2b2477b617f94a5ccd8e687661f134fb4d0f1732e1d8fd050498a199","observation_id":"76374e6f-e93d-4173-9cf9-21e7ab680101","resolution":{"observed_at":"2026-08-06T19:43:00.439013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.427021Z","title":"Data set: Eigenmike-DRIRs, KEMAR 45BA-BRIRs, RIRs and 360° pictures captured at five positions of a small conference room,","venue":null,"work_id":"eb455ebd-cc1a-408c-a1da-44977376b6d4","year":2019},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.160983Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:819af1a70fc11f99136a112ad1bb1b92c000047f7ea0bf546c6f28c2d0d229b9","observation_id":"61eb1f6e-e971-4863-bcef-f8adff76a778","resolution":{"observed_at":"2026-08-06T19:43:00.430256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02988","last_updated":"2025-04-03T19:27:43Z","snapshot_observed_at":"2026-08-16T12:44:08.995884Z","submitted_at":"2025-04-03T19:27:43Z","title":"Generating Diverse Audio-Visual 360 Soundscapes for Sound Event Localization and Detection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02988","snapshot_observed_at":"2026-08-06T19:43:00.163799Z","title":"Generating di- verse audio-visual 360 soundscapes for sound event localization and detection,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.163799Z"},"links":{"cited_paper":"/paper/2504.02988","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:1699f4beb1a9dbf2aaf54c5ffba57ae57d596309a8fda0d1d89af53a83fb8539","observation_id":"bd94749f-6952-4e7d-b5b2-c6df168b3df1","resolution":{"observed_at":"2026-08-06T19:43:00.163799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.418385Z","title":"Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models,","venue":null,"work_id":"a9d0d7cb-8c41-4cc7-b727-5d8ca306da3a","year":2015},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.167856Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:6b0f180946c48a0d909c5e5fda82a145fca2f6a251d913d1c9354c9f00e127a6","observation_id":"773fc60b-4f9c-45aa-9edc-134a438af119","resolution":{"observed_at":"2026-08-06T19:43:00.421430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.408798Z","title":"Robust and adaptive door operation with a mobile robot,","venue":null,"work_id":"eab234bb-6f81-4250-8fbd-dc86d2c495db","year":2021},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.170687Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:2d519f01ec7e3d061947d9c7b3795d6cc08394f1caed9547611bc1cd3f20e563","observation_id":"2e79874e-6267-4817-9bf4-45bec38bb158","resolution":{"observed_at":"2026-08-06T19:43:00.412566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02030","last_updated":"2024-12-06T11:22:17Z","snapshot_observed_at":"2026-08-17T21:27:31.930644Z","submitted_at":"2024-12-02T23:20:35Z","title":"NitroFusion: High-Fidelity Single-Step Diffusion through Dynamic Adversarial Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02030","snapshot_observed_at":"2026-08-06T19:43:00.173223Z","title":"NitroFusion: High-fidelity single-step diffusion through dynamic adversarial training,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.173223Z"},"links":{"cited_paper":"/paper/2412.02030","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:d1d915203d1104c7fea28b187e9486ee6e8ecf23e0750ddaf3c9843557347ace","observation_id":"ab8944cd-4ebf-491c-a93b-75ad25f951a0","resolution":{"observed_at":"2026-08-06T19:43:00.173223Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.399884Z","title":"360-Indoor: Towards learning real-world objects in 360° indoor equirectangular images,","venue":null,"work_id":"faba7fb5-4c1c-48f8-be21-9afb9b85c256","year":2020},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.176261Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:0087b9130996861ccf74639fde64a985717086dcf5b168f410d6b87b1408e4d8","observation_id":"31578a05-2ae2-4a00-84b7-60cdc714dda5","resolution":{"observed_at":"2026-08-06T19:43:00.403173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.390325Z","title":"Exploring audio-visual information fusion for sound event localization and detection in low-resource realistic scenarios,","venue":null,"work_id":"8172af58-c70f-42c2-83aa-414352ffb383","year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.178838Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:7da6715938475e42e57d8f2b93a61afb2f8120d335e361a620e523bfb2988563","observation_id":"3eb7450e-8f09-421f-94e4-b9928b7e6809","resolution":{"observed_at":"2026-08-06T19:43:00.394140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17129","last_updated":"2024-01-29T06:05:23Z","snapshot_observed_at":"2026-08-16T14:23:46.827475Z","submitted_at":"2024-01-29T06:05:23Z","title":"Enhanced Sound Event Localization and Detection in Real 360-degree audio-visual soundscapes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17129","snapshot_observed_at":"2026-08-06T19:43:00.181477Z","title":"Enhanced sound event localization and de- tection in real 360-degree audio-visual soundscapes,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.181477Z"},"links":{"cited_paper":"/paper/2401.17129","citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:9ac1bc96ece6166606dd094c4c4bfa00137a506eec93918d8961ed484cd4908c","observation_id":"62e96e29-627e-41a5-929a-ac74af12b4d5","resolution":{"observed_at":"2026-08-06T19:43:00.181477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T19:43:00.381891Z","title":"Ultralytics YOLO,","venue":null,"work_id":"26248ec9-7de3-4e8e-b031-61668edfa719","year":2025},"citing_paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:00.184577Z"},"links":{"citing_paper":"/paper/2507.04845"},"observation_digest":"sha256:3cff7f9fbd1d8942aaf3a7c258993ea2484d0a7febf158d9a8d71e310c24545e","observation_id":"78a7b421-e9a3-415d-888f-4c7c84bafd09","resolution":{"observed_at":"2026-08-06T19:43:00.384730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.04845","last_updated":"2025-07-07T10:08:57Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-17T19:33:52.396398Z","submitted_at":"2025-07-07T10:08:57Z","title":"Spatial and Semantic Embedding Integration for Stereo Sound Event Localization and Detection in Regular Videos"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":3,"verified_fuzzy":40},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 0 inbound Pith citation observations for arXiv:2507.04845."}