{"as_of":"2026-08-22T14:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:597d0353b7e90574da5d6531e43ebcf4c78b7a2bacbb28dcf0931134e576a813","coverage":[{"denominator":51,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":51,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T21:55:00.437900Z","state":"measured"},{"denominator":51,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":51,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.07829/citation-record","integrity":"/paper/2508.07829/integrity","json":"/paper/2508.07829/citation-record.json","paper":"/paper/2508.07829"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T21:54:55.018861Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.018861Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:47b1270d758d26ee868f9bbc6d08c65170b1c487786424e55bad6bfeb41e4d89","observation_id":"d799a26c-be9e-4a90-bdc0-004f4d199ce9","resolution":{"observed_at":"2026-08-05T21:54:55.018861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-05T21:54:55.093314Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.093314Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:fd4163b325816da10af0e6b3b4edebc774343f97c570900a24a67963f5df8ced","observation_id":"4e1cdf58-475e-45f1-b332-78498679b8c4","resolution":{"observed_at":"2026-08-05T21:54:55.093314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-08-18T18:18:37.449517Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-05T21:54:55.207940Z","title":"Deepseek-v3 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.207940Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:9d0040dc48d750ec0c910931693058a0886059d08995861a0a0294d20d987779","observation_id":"24a23040-184e-4408-8fb7-4eb7dd8f6171","resolution":{"observed_at":"2026-08-05T21:54:55.207940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:09.594388Z","title":"Virtanen, M","venue":null,"work_id":"d0b4bf96-f7c0-44c8-ab0f-a6814f2e479d","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.290539Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:6a5a2febd16b72b162f5f067d7b07effcbaae63dba9a1872ecf82986296c22e1","observation_id":"706dea90-ca0a-4a26-8799-cf7c8e25b31a","resolution":{"observed_at":"2026-08-05T21:55:09.763230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:09.308624Z","title":"Sound event detection in domestic environments with weakly labeled data and soundscape synthesis,","venue":null,"work_id":"f1920bf5-0b41-47ec-8938-2369a5c41ae5","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.368526Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:c40513d7dc0895d2759ef3c821c021469bad821672fa68687e33c6d3be4fe1e7","observation_id":"fef540cf-73ea-4132-a74f-d3f7504bc6b3","resolution":{"observed_at":"2026-08-05T21:55:09.449372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.996537Z","title":"Convolutional recurrent neural networks for polyphonic sound event detection,","venue":null,"work_id":"a63992ca-be84-4383-bead-1fd46ee3fe1b","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.459643Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:714364f484f2a05b1915accfcb36cfcd8904395761bdb975f75e2de486f4bec9","observation_id":"5bd1999d-f935-467f-9f97-9e972676f921","resolution":{"observed_at":"2026-08-05T21:55:09.138402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.617758Z","title":"Metrics for polyphonic sound event detection,","venue":null,"work_id":"debff7bb-316a-4114-9456-3cb80657e479","year":2016},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.545487Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:8df5fc740cc0b6ec1ea0828e671a3eafa085f7a167db2a3f851a10d6abe8c655","observation_id":"eb209ae2-e210-4867-93c9-415fbddc20b0","resolution":{"observed_at":"2026-08-05T21:55:08.824581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:08.274968Z","title":"A framework for the robust evaluation of sound event detection,","venue":null,"work_id":"4d7b2e4a-23ad-4e41-a9aa-a43770daa170","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.616813Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:8d039f4c459aa1e816eaf0d322018cf14ceeb8fec786a5413cec5dd1d413f199","observation_id":"230cba28-708a-4c03-b05e-054c3d08235c","resolution":{"observed_at":"2026-08-05T21:55:08.447221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.07208","last_updated":"2025-08-27T08:09:06Z","snapshot_observed_at":"2026-08-17T02:40:18.869178Z","submitted_at":"2025-02-11T03:07:03Z","title":"Towards Understanding of Frequency Dependence on Sound Event Detection","version":2},"cited_work":{"arxiv_id":"2502.07208","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.07208","snapshot_observed_at":"2026-08-05T21:55:02.086940Z","title":"Towards Understanding of Frequency Dependence on Sound Event Detection","venue":"eess.AS","work_id":"e2e70295-dd51-4328-868c-5b31e026f5a7","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.618843Z"},"links":{"cited_paper":"/paper/2502.07208","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:77185ff67958285d0e9118ec0cddea36dc968b6af6879a53c878ef848abc8613","observation_id":"9aa04b05-3827-4cbe-a996-4e467a4ee220","resolution":{"observed_at":"2026-08-05T21:55:02.185813Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.20857","last_updated":"2025-02-28T08:55:20Z","snapshot_observed_at":"2026-08-19T06:26:18.551536Z","submitted_at":"2025-02-28T08:55:20Z","title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2502.20857","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.20857","snapshot_observed_at":"2026-08-05T21:55:01.925774Z","title":"JiTTER: Jigsaw Temporal Transformer for Event Reconstruction for Self-Supervised Sound Event Detection","venue":"eess.AS","work_id":"adee432d-4163-4ef8-ab79-dc7ca5f391eb","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.621227Z"},"links":{"cited_paper":"/paper/2502.20857","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:94449723879ebe7bb9cd82d7d2bb231d2f1d1ee8680ef8cc1ad838fac34da2f7","observation_id":"b63d5103-2042-45f4-a322-ae8232c20e11","resolution":{"observed_at":"2026-08-05T21:55:02.001185Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.980508Z","title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition,","venue":null,"work_id":"1c8c02e6-692c-450a-b035-d3922cd1b210","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.623704Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:d3fe3f5cf2186070f0c196ffb8163e849db1bc135130e2289fb17d731c2b473f","observation_id":"5af3869f-a693-4e89-8801-49d2dc06c548","resolution":{"observed_at":"2026-08-05T21:55:08.124515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.715689Z","title":"Conformer: Convolution- augmented Transformer for Speech Recognition,","venue":null,"work_id":"61a3b71a-006e-442d-9318-a3389e4ca984","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.696331Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:5fac05241aa920eff275c2cad99d45dcce6fbda75414d33152a816a0ae3a94ed","observation_id":"cd7fd502-ae51-4374-af28-ea291568350d","resolution":{"observed_at":"2026-08-05T21:55:07.854378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.433681Z","title":"Coherence-based phonemic analysis on the ef- fect of reverberation to practical automatic speech recognition,","venue":null,"work_id":"28b1b0d5-64c4-4ecd-9ed7-b6f7fd63852a","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.792917Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:92c177fcc5956ec7e1b4eae83823c05c4c2296ad454b3157717709d30e236da6","observation_id":"e6ce8a0c-c5a9-4e18-8add-d94a0a1e06a2","resolution":{"observed_at":"2026-08-05T21:55:07.597433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:07.130046Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representations,","venue":null,"work_id":"0aaa0871-ab2b-472c-8468-1e852cf6c7f3","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.856904Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:accbb674f129c19be113f6b4e9f83cce3133b4fe2337c5279b5f658f13679d0c","observation_id":"9e209df9-6509-4644-808f-04218b52ba55","resolution":{"observed_at":"2026-08-05T21:55:07.258853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.901570Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"cc1e6e9a-c708-4f2f-9352-4cce52039ce8","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:55.890183Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:04c523d7234139ae91afa107eca37775fd6911347690d6872330aaad6ec8c5c2","observation_id":"8f743f9e-7593-4972-969e-e9173af92efa","resolution":{"observed_at":"2026-08-05T21:55:07.011330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:54:56.043026Z","title":"Attentive statistics pooling for deep speaker embedding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.043026Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:4f75da868b14048b11ecbd9fb2637a7d163239f58d110d974a7648eede7cde0d","observation_id":"60b67069-b919-4e57-8af0-b7ef5baa3df6","resolution":{"observed_at":"2026-08-05T21:54:56.043026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.758910Z","title":"Exploring the encoding layer and loss function in end-to-end speaker and language recognition system,","venue":null,"work_id":"f3d30652-3146-4407-8eb0-5781de8212dc","year":2018},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.245420Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:caab03d64be974e7fb4dbac466d3ae456d9a3d9632522a9166bff77dbd52b025","observation_id":"68704b73-a938-4901-9684-5e008bac89d6","resolution":{"observed_at":"2026-08-05T21:55:06.780807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.649259Z","title":"Analysis-based optimization of temporal dynamic convolutional neural network for text-independent speaker verification,","venue":null,"work_id":"1512300f-c34f-436d-8297-56747a9d0949","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.417343Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:214f04dfc38f19a51b3e41fe50887a3c50e585c890b1deb4cfe3166b1e6a0a16","observation_id":"188f8834-4afc-4ca9-bfd0-63c267134a61","resolution":{"observed_at":"2026-08-05T21:55:06.713473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.538687Z","title":"Integrating fre- quency translational invariance in tdnns and frequency positional in- formation in 2d resnets to enhance speaker verification,","venue":null,"work_id":"48f62d3f-3788-4093-99e8-79cbea17f837","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.581826Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7618c0731587f42da438a969a44cb9caf755c5c8ecd85027e5c31066efaf9b35","observation_id":"e88911b6-715e-4cb2-abf6-f55cbece2c99","resolution":{"observed_at":"2026-08-05T21:55:06.589708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.381522Z","title":"Convolution-based channel-frequency atten- tion for text-independent speaker verification,","venue":null,"work_id":"3afefbe0-dbd8-446d-a2f4-c8cd0c60871a","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.697963Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:123a205a4c7e53a7e07bda450f1b457356c42c15458b731208b760728a688035","observation_id":"f6828fbc-3eea-4ec9-858f-b650999147e0","resolution":{"observed_at":"2026-08-05T21:55:06.462490Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.227720Z","title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition,","venue":null,"work_id":"36783fe0-2da0-4329-97da-a1deb5730e0e","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:56.862622Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:8bdc78a80c802432df0a36c20e5d43ef6dc4d56cfb8db7ab8723f717d9c0692c","observation_id":"b0868592-5570-465e-a0d1-f6f08c9e6e85","resolution":{"observed_at":"2026-08-05T21:55:06.313111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:06.037049Z","title":"Deep learning based cough detection camera using enhanced features,","venue":null,"work_id":"2c451e9b-529b-41d6-8fd8-a798c994ec87","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.030581Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:57b0f02f368e8d5bed2860dd5e293da8865d93d1170d0b46ba2924490181ee54","observation_id":"2de542e1-234c-4a89-9208-af40a6aa3584","resolution":{"observed_at":"2026-08-05T21:55:06.135508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.806716Z","title":"Real-time sound recognition system for human care robot considering custom sound events,","venue":null,"work_id":"e4eaf4f8-a38a-4a82-a9bd-007adf7354ea","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.193849Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:97cc42cc6b5f3f7ec08c032aaa86b2def5f88673dfb32eb97781b15a5017c38c","observation_id":"d64d0547-8f1c-40a6-8018-c7acb8450811","resolution":{"observed_at":"2026-08-05T21:55:05.927205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.631041Z","title":"Ast: Audio spectrogram trans- former,","venue":null,"work_id":"83d3fe6e-598f-4446-8882-33726e6e6bfb","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.356267Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:88de57e57e4a2ed12e2ab24584bf7492f35c10035e2f70cc323334977d23edf5","observation_id":"f3480612-f4b5-44cb-815d-fbe705c16e93","resolution":{"observed_at":"2026-08-05T21:55:05.744457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.490136Z","title":"Beats: Audio pre-training with acoustic tokenizers,","venue":null,"work_id":"3eeeba9c-0ad6-4143-84a9-5e9782191805","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.480436Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:65cf9f181d9357984cf8b53308f3236a7048618a47025c95c4f558436dba8dce","observation_id":"049bd74c-97f3-445e-8e5c-83cf7d744dfe","resolution":{"observed_at":"2026-08-05T21:55:05.563083Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.315692Z","title":"Overview and evaluation of sound event localization and detection in dcase 2019,","venue":null,"work_id":"2cf25465-590c-451f-bf76-dbfbb4752922","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.627088Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:ab1f3bd12eedabb62d83afa1027182eb2a1ff57509fd0cea7a48d3d59aac424d","observation_id":"a325f9c7-30a8-497f-9c02-cd21b79416e7","resolution":{"observed_at":"2026-08-05T21:55:05.397189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:05.143613Z","title":"STARSS22: A dataset of spatial recordings of real scenes with spa- tiotemporal annotations of sound events,","venue":null,"work_id":"ee6f6c60-ff04-4bd4-98b6-75bde9420fb2","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.738938Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:31513ce458006e9a310d66dcffd17064fe1386fbe22beae6d07a52c54757deff","observation_id":"02947e24-3dad-417e-89fe-c00cdcf9eba2","resolution":{"observed_at":"2026-08-05T21:55:05.229253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.964032Z","title":"Data augmentation and squeeze-and-excitation network on multiple dimension for sound event localization and detection in real scenes,","venue":null,"work_id":"0dcef2bd-e777-4a2c-91d9-968028dde5d2","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:57.899374Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:09ad08161b19c86f02b3a9994fef80cc8a3309ae75fd1a765f729ccef9673af7","observation_id":"9e0b3c59-c1e2-4daa-bfd4-278b4d3cd8c5","resolution":{"observed_at":"2026-08-05T21:55:05.050564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20530","last_updated":"2025-07-28T05:27:07Z","snapshot_observed_at":"2026-08-18T21:43:44.793607Z","submitted_at":"2025-07-28T05:27:07Z","title":"Binaural Sound Event Localization and Detection based on HRTF Cues for Humanoid Robots","version":1},"cited_work":{"arxiv_id":"2507.20530","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.20530","snapshot_observed_at":"2026-08-05T21:55:01.738106Z","title":"Binaural Sound Event Localization and Detection based on HRTF Cues for Humanoid Robots","venue":"eess.AS","work_id":"a19684ed-d865-4b12-bc42-06f159a5eec5","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.024131Z"},"links":{"cited_paper":"/paper/2507.20530","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:4701024f629fa4e7fe973eebbcc811b1ea297fd8afe33288176913f8c6362995","observation_id":"30625b5f-9cdb-4af7-b716-9cdc483e3830","resolution":{"observed_at":"2026-08-05T21:55:01.847274Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.793104Z","title":"Automated audio captioning with recurrent neural networks,","venue":null,"work_id":"115955b1-8091-4840-ae3b-aa9f0b3f0750","year":2017},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.170275Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:e28096d0e79c3cb636a681fb36e5bc68afad8f70c213889c6a8aae15005502f1","observation_id":"1e23078e-4c6b-4e27-8ea9-b43acbe34508","resolution":{"observed_at":"2026-08-05T21:55:04.881036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.634115Z","title":"Clotho: an audio captioning dataset,","venue":null,"work_id":"bff492f5-9e07-404c-ab73-0bf2fc1422bd","year":2020},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.286059Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:ca0abe92e71afd4c3fe629120e31fbd0408318716b8cbfb9f687c13addbd194c","observation_id":"1a0665ed-0268-45ac-b880-3125acd53a1d","resolution":{"observed_at":"2026-08-05T21:55:04.676532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.448400Z","title":"Chatgpt caption paraphrasing and fense-based caption filtering for automated audio captioning,","venue":null,"work_id":"cbc06dc6-e05d-4911-963c-cc58abf91e1a","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.404359Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7bff6b03d7ab2d758a294f84fac74423ca66bfad5660b3fa4b54d00c0c316969","observation_id":"77818c72-35bb-4d60-bb00-17b96f46a42f","resolution":{"observed_at":"2026-08-05T21:55:04.531602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18638","last_updated":"2024-03-27T14:44:24Z","snapshot_observed_at":"2026-08-16T14:05:51.760460Z","submitted_at":"2024-03-27T14:44:24Z","title":"Mind the Domain Gap: a Systematic Analysis on Bioacoustic Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2403.18638","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.18638","snapshot_observed_at":"2026-08-05T21:55:01.548902Z","title":"Mind the Domain Gap: a Systematic Analysis on Bioacoustic Sound Event Detection","venue":"eess.AS","work_id":"6fecad86-b2fd-474a-bd65-6a83f0f20902","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.488390Z"},"links":{"cited_paper":"/paper/2403.18638","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:56f486da570f7c3c12a7f4262545a06dfb9347c6e5702880a050bb4930567587","observation_id":"556bbac1-d0bc-4baf-a758-c59642c58474","resolution":{"observed_at":"2026-08-05T21:55:01.629069Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.305757Z","title":"Few-shot bioacoustic event detection utilizing spectro-temporal receptive field,","venue":null,"work_id":"2282fd05-985d-4d9c-8274-d4c2ef098e0f","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.572096Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:16e76784b89884ffddffdfb4887fc087f1949739ddc24ba0944a5fa261cd001e","observation_id":"9633a1d8-64e0-48fd-bcfc-fabe39d98b4f","resolution":{"observed_at":"2026-08-05T21:55:04.395425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:04.171737Z","title":"Prtfnet: Hrtf individual- ization for accurate spectral cues using a compact prtf,","venue":null,"work_id":"e3a6651a-791f-4d28-910b-350d2dd37ef8","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.674078Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:640f0674eb3a0d231fb6552a60dc79a8bf1b7549feeeb9dd00e196cdd3f52124","observation_id":"a613f9fe-77bb-4d9b-b2a5-ca500286eb07","resolution":{"observed_at":"2026-08-05T21:55:04.245019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.991299Z","title":"Filteraugment: An acoustic environmental data augmentation method,","venue":null,"work_id":"2ab2deb3-d288-4223-bf2c-0cf1fa52b99b","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.787521Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:b0ece0dffee83ca18e7cc6a0e5d515b7a17503c9ff41f2172f765eee36af0ccd","observation_id":"d6f13d25-fadb-4cf4-90c1-cdf14ff29460","resolution":{"observed_at":"2026-08-05T21:55:04.073974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14817","last_updated":"2025-04-21T02:50:34Z","snapshot_observed_at":"2026-08-22T09:31:28.379230Z","submitted_at":"2025-04-21T02:50:34Z","title":"DNN based HRIRs Identification with a Continuously Rotating Speaker Array","version":1},"cited_work":{"arxiv_id":"2504.14817","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.14817","snapshot_observed_at":"2026-08-05T21:55:01.314634Z","title":"DNN based HRIRs Identification with a Continuously Rotating Speaker Array","venue":"eess.AS","work_id":"d5183c36-f9fc-4b06-b7c6-8459785720f9","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.901523Z"},"links":{"cited_paper":"/paper/2504.14817","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:02e1dc0b418b872d081a2fc3300e89868df1829b8aa61f92c3f6bdaee21b0355","observation_id":"369327be-28db-4ef5-baf5-9b2f03d07e45","resolution":{"observed_at":"2026-08-05T21:55:01.438903Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.845478Z","title":"AudioLDM: Text-to-audio generation with latent diffusion models,","venue":null,"work_id":"cd01c446-6730-492d-b9ad-f7296d2e1859","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:58.988369Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:09f63b52d2768c5ca16796c247c09257917971674498b38119e053c4fdca7957","observation_id":"792542b4-877c-4b7c-922c-55834e3b1dff","resolution":{"observed_at":"2026-08-05T21:55:03.901509Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.688357Z","title":"Audiogen: Textually guided audio generation,","venue":null,"work_id":"1adec640-b89e-48ed-bea7-31040c9fbb05","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.102365Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:84e75dedc2235b8a7f1fe47cad877bbc38e134a53e99c59a7179fc82a8beec80","observation_id":"ada6fa31-0d43-4223-a3dc-defab0890554","resolution":{"observed_at":"2026-08-05T21:55:03.753031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.478428Z","title":"Vifs: An end-to-end variational inference for foley sound synthesis,","venue":null,"work_id":"468c9414-08b1-4101-8c41-b79a02f13f51","year":2023},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.215508Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:9a9c7307093a8f11d776772a9cd4f1ac053fc3777dcb682930684b3ca698378c","observation_id":"d7c109d0-d67f-4729-bc6d-1c51d4b0c51b","resolution":{"observed_at":"2026-08-05T21:55:03.554341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.341487Z","title":"Heavily augmented sound event detection utilizing weak predictions,","venue":null,"work_id":"caf02097-7ece-45b4-b53d-87de3b904d47","year":2021},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.349962Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:4321392ac2070b35f868197e7b1f1e772bf18f3e89c240a22ff434ffa8d368bb","observation_id":"7bcbe46f-95b9-49f2-91b4-e1d3c2899ae0","resolution":{"observed_at":"2026-08-05T21:55:03.411377Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:03.193229Z","title":"Filteraugment: An acoustic environmental data augmentation method,","venue":null,"work_id":"096cd250-ff3d-49e5-a112-d65df3218627","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.472133Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:a72e25fd2b80b359f8761615324aa42427c8cd6be3be77a06d88ef7c65628480","observation_id":"21a5f27c-e399-4348-a384-1d74fc22a414","resolution":{"observed_at":"2026-08-05T21:55:03.264445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.984671Z","title":"Frequency Dynamic Convolution: Frequency-Adaptive Pattern Recognition for Sound Event Detection,","venue":null,"work_id":"1ed47e4e-a41a-4b71-be56-c8cd5924667b","year":2022},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.555478Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:5ef8e008cd536414675ac37caf39409abe23ec3541ef5a508313cb6cd087865e","observation_id":"1973ca4d-8cb7-4cb9-a7b5-7759246abec0","resolution":{"observed_at":"2026-08-05T21:55:03.075348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.828505Z","title":"Self training and ensembling frequency dependent networks with coarse prediction pooling and sound event bounding boxes,","venue":null,"work_id":"b13e2f6e-28a0-4401-8bcc-3fc353a5d39f","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.654198Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:0da4b692a343eefc694ac90c76606af6725f4c662360de946d64b0aa8f646c47","observation_id":"ded12c34-a415-43b6-9e1d-3e0d9102ebec","resolution":{"observed_at":"2026-08-05T21:55:02.909992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.582196Z","title":"Diversifying and expanding frequency-adaptive convolution kernels for sound event detection,","venue":null,"work_id":"e43dd54a-9390-4390-beeb-d757ebe1723e","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.795404Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:eb273a5514ab4a63c5501a4c007a45232be73df5c0cf4629e175d49b509c0cbe","observation_id":"0ae9c7db-d015-4fed-a50c-4e1f603a90e2","resolution":{"observed_at":"2026-08-05T21:55:02.747332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13312","last_updated":"2024-09-20T02:18:47Z","snapshot_observed_at":"2026-08-16T13:41:33.641896Z","submitted_at":"2024-06-19T08:02:02Z","title":"Pushing the Limit of Sound Event Detection with Multi-Dilated Frequency Dynamic Convolution","version":3},"cited_work":{"arxiv_id":"2406.13312","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.13312","snapshot_observed_at":"2026-08-05T21:55:01.157686Z","title":"Pushing the Limit of Sound Event Detection with Multi-Dilated Frequency Dynamic Convolution","venue":"eess.AS","work_id":"ce60f310-b3b1-4276-842a-8b1e114f638d","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.894205Z"},"links":{"cited_paper":"/paper/2406.13312","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:404b5c84e2b5ba9c3f87ddc394630fec45196bccd417311c8f8498e2ffaff5d0","observation_id":"05d450ad-e4a7-4d4c-820a-035c26b9ba0f","resolution":{"observed_at":"2026-08-05T21:55:01.245826Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12670","last_updated":"2025-04-17T06:03:43Z","snapshot_observed_at":"2026-08-16T12:23:17.817807Z","submitted_at":"2025-04-17T06:03:43Z","title":"Temporal Attention Pooling for Frequency Dynamic Convolution in Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2504.12670","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.12670","snapshot_observed_at":"2026-08-05T21:55:00.961516Z","title":"Temporal Attention Pooling for Frequency Dynamic Convolution in Sound Event Detection","venue":"eess.AS","work_id":"38288718-f4f7-440a-972a-57ea6f3fb17f","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-05T21:54:59.989934Z"},"links":{"cited_paper":"/paper/2504.12670","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:73127fc8c04fb2f48ab0b0e115dad9c8e3f6665e85bc689334b360d63d2f5ba9","observation_id":"72f54b94-765d-44ca-ac9e-ddc6677757af","resolution":{"observed_at":"2026-08-05T21:55:01.067562Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12785","last_updated":"2025-06-15T09:32:16Z","snapshot_observed_at":"2026-08-07T00:40:07.029946Z","submitted_at":"2025-06-15T09:32:16Z","title":"Frequency Dynamic Convolutions for Sound Event Detection","version":1},"cited_work":{"arxiv_id":"2506.12785","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.12785","snapshot_observed_at":"2026-08-05T21:55:00.763800Z","title":"Frequency Dynamic Convolutions for Sound Event Detection","venue":"eess.AS","work_id":"dc43963d-3864-4bf9-bec7-a9e062074257","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.075447Z"},"links":{"cited_paper":"/paper/2506.12785","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:7392c5aabcaa2c8daf808db357b237206431b97a189356aa8a8f00679e567bbd","observation_id":"773b6fc6-25d1-44e5-86c6-7f4ef1b8a7d8","resolution":{"observed_at":"2026-08-05T21:55:00.851246Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05335","last_updated":"2025-06-08T18:21:26Z","snapshot_observed_at":"2026-08-21T12:51:56.879148Z","submitted_at":"2025-05-08T15:27:43Z","title":"FLAM: Frame-Wise Language-Audio Modeling","version":2},"cited_work":{"arxiv_id":"2505.05335","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.05335","snapshot_observed_at":"2026-08-05T21:55:00.563579Z","title":"FLAM: Frame-Wise Language-Audio Modeling","venue":"cs.SD","work_id":"547395b7-81eb-43ab-8ba3-6f1d93afb6ff","year":2025},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.164281Z"},"links":{"cited_paper":"/paper/2505.05335","citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:df03b7c9f67fd2288a6edbf265c20b9a3516b793102fc29c25b4dec6386c2b42","observation_id":"f3f7fd21-81b5-464d-82dc-1d6b272b7750","resolution":{"observed_at":"2026-08-05T21:55:00.654739Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.421355Z","title":"AudioCaps: Generating captions for audios in the wild,","venue":null,"work_id":"a4e289d4-951a-474a-a3a9-82073007ae4a","year":2019},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.299272Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:0c24be4215587541d1218eb8efa6569efe9bb5a1bf1c91ea040a1ca4705b2818","observation_id":"2d676abf-5417-479d-85b2-bcc09366cf9a","resolution":{"observed_at":"2026-08-05T21:55:02.500837Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T21:55:02.290136Z","title":"Wavcaps: A chatgpt-assisted weakly-labelled audio captioning dataset for audio-language multimodal research,","venue":null,"work_id":"6ec90f8a-8b80-49d7-9c41-46b8f09bfd19","year":2024},"citing_paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-05T21:55:00.437900Z"},"links":{"citing_paper":"/paper/2508.07829"},"observation_digest":"sha256:c9a37dd2d8d95adf1fb78577be5e4754225e241b16066f32b0c3edbe7c4586ae","observation_id":"476396e1-5fa4-4778-a4b8-db33cee5beeb","resolution":{"observed_at":"2026-08-05T21:55:02.344444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.07829","last_updated":"2025-08-11T10:25:58Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-21T12:52:12.200460Z","submitted_at":"2025-08-11T10:25:58Z","title":"Auditory Intelligence: Understanding the World Through Sound"},"reference_resolution":{"displayed":51,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":4,"verified_exact":9,"verified_fuzzy":38},"total_outbound_references":51},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 51 of 51 outbound references and 0 inbound Pith citation observations for arXiv:2508.07829."}