{"as_of":"2026-08-14T23:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f59e8ad3130e90cae7ba8e6565a8fa6e1c0a769ddb3e6768cf3d5ee17c7f271e","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:21:50.792032Z","state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:18:51.939607Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T11:49:50.788200Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T19:21:26.933349Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2504.18425"},"observation_digest":"sha256:93474420664a71c4f68c1e629710f3b7d44bc4e8eaabf8538d4966548ddbfde2","observation_id":"203b544f-3836-4423-a139-4e5a5f95024d","resolution":{"observed_at":"2026-05-11T19:21:27.283015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-07T10:18:51.939607Z","title":"Osum: Advancing open speech un- derstanding models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05796","last_updated":"2025-06-06T06:43:34Z","snapshot_observed_at":"2026-08-11T19:39:02.266479Z","submitted_at":"2025-06-06T06:43:34Z","title":"Diarization-Aware Multi-Speaker Automatic Speech Recognition via Large Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:51.939607Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2506.05796"},"observation_digest":"sha256:22f3af362dcb96a8fc6b2a69caa061c68e1c72b9d53cb2b28cfd06ab0c44afe0","observation_id":"314f9479-bc79-4f09-bcb5-5fa240649b68","resolution":{"observed_at":"2026-08-07T10:18:51.939607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-05T20:59:46.041859Z","title":"What do you think I should eat?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09599","last_updated":"2026-05-23T09:18:10Z","snapshot_observed_at":"2026-08-07T16:58:22.699012Z","submitted_at":"2025-08-13T08:28:21Z","title":"BridgeTA: Bridging the Representation Gap in Knowledge Distillation via Teacher Assistant for Bird's Eye View Map Segmentation","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-05T20:59:46.041859Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2508.09599"},"observation_digest":"sha256:1dc95b9617b943c8717e8b24c47f8af733ec6af4096f4501cc0ba8f36f99780c","observation_id":"27882c48-9fd5-4dce-80ee-c7519ae63114","resolution":{"observed_at":"2026-08-05T20:59:46.041859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-05T21:03:09.640862Z","title":"What do you think I should eat?","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.09600","last_updated":"2025-09-03T13:33:34Z","snapshot_observed_at":"2026-08-10T11:32:29.120466Z","submitted_at":"2025-08-13T08:30:14Z","title":"OSUM-EChat: Enhancing End-to-End Empathetic Spoken Chatbot via Understanding-Driven Spoken Dialogue","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-05T21:03:09.640862Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2508.09600"},"observation_digest":"sha256:4f0b8fec45e006db01b0a71cf19a364db4188a8e5c82ae35a9cbe06b2e80b926","observation_id":"787c8920-5478-449b-bb67-7791784322b4","resolution":{"observed_at":"2026-08-05T21:03:09.640862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2509.14804","last_updated":"2025-09-18T09:59:55Z","snapshot_observed_at":"2026-08-07T14:31:44.422604Z","submitted_at":"2025-09-18T09:59:55Z","title":"Towards Building Speech Large Language Models for Multitask Understanding in Low-Resource Languages","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T16:27:37.596817Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2509.14804"},"observation_digest":"sha256:cdb2b5aa51947860a9da1cc7cd957c72887fc576d0aaac41cd4c108e7056244e","observation_id":"93284ac9-394a-42dc-9ced-1d4b75bc47dd","resolution":{"observed_at":"2026-05-18T16:31:37.228057Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-03T17:16:38.479016Z","title":"Osum: Advancing open speech understanding models with limited resources in academia","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.10324","last_updated":"2026-06-22T06:41:41Z","snapshot_observed_at":"2026-08-03T17:16:35.340426Z","submitted_at":"2025-12-11T06:18:58Z","title":"EchoingPixels: Aliasing-Resistant Joint Token Reduction for Audio-Visual LLMs","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T17:16:38.479016Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2512.10324"},"observation_digest":"sha256:2c110d7678182ebd416543ac641605af4aac568863598f7b21c3f45db17a470a","observation_id":"267ba0df-de4f-406e-956d-5b4178a6c813","resolution":{"observed_at":"2026-08-03T17:16:38.479016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.11594","last_updated":"2026-04-24T06:36:24Z","snapshot_observed_at":"2026-08-14T21:13:08.485449Z","submitted_at":"2026-04-13T15:06:05Z","title":"HumDial-EIBench: A Human-Recorded Multi-Turn Emotional Intelligence Benchmark for Audio Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T15:24:27.118694Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.11594"},"observation_digest":"sha256:4096c628959dea455f20a7b58e17e15caf8aabb174bc1cb5fef4630bf3844e40","observation_id":"196a7acf-ff2b-4f85-a4b7-7f2990d536e4","resolution":{"observed_at":"2026-05-11T10:36:05.898403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T14:10:03.707886Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:acfd24152e7fd26213f6283f440ab8cbd8632e0e23fac026986a48493a323051","observation_id":"6ead0d2e-a824-4f9f-951d-49a72e4765b3","resolution":{"observed_at":"2026-05-10T14:10:28.342560Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-12T21:18:46.566338Z","title":"Osum: Advancing open speech understand- ing models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.12527","last_updated":"2026-07-03T08:49:19Z","snapshot_observed_at":"2026-07-12T21:18:44.451500Z","submitted_at":"2026-04-14T10:00:39Z","title":"Audio-Cogito: Towards Deep Audio Reasoning in Large Audio Language Models","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T21:18:46.566338Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.12527"},"observation_digest":"sha256:aa5c955f7a27dd461dd32e216c8e62876f70492ad06b2060f4d21165a1588b53","observation_id":"d2db76cc-80e9-4dfd-93d8-fbf512f19a12","resolution":{"observed_at":"2026-07-12T21:18:46.566338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2604.18204","last_updated":"2026-04-20T12:54:14Z","snapshot_observed_at":"2026-07-06T23:05:08.728446Z","submitted_at":"2026-04-20T12:54:14Z","title":"Hard to Be Heard: Phoneme-Level ASR Analysis of Phonologically Complex, Low-Resource Endangered Languages","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T04:25:37.725216Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2604.18204"},"observation_digest":"sha256:26345af2e22ebf4b711d62022fa0f1b6d4821679112119ded38a25a1b211c096","observation_id":"ae99d2f1-9b81-426c-9ce1-96113b53d4bc","resolution":{"observed_at":"2026-05-11T11:56:31.006517Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":"2501.13306","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-07-04T11:49:50.788200Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","venue":null,"work_id":"45f98848-4dc2-4c34-9aac-786ca4cd4ace","year":2025},"citing_paper":{"arxiv_id":"2606.22868","last_updated":"2026-06-22T05:24:35Z","snapshot_observed_at":"2026-08-14T11:36:02.442036Z","submitted_at":"2026-06-22T05:24:35Z","title":"MSU-Bench: Towards Speaker-Centric Understanding in Conversational Multi-Speaker Scenarios","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-26T07:36:34.307652Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2606.22868"},"observation_digest":"sha256:22e90509c675e1c9d753fa0fa28d2d1327ac96476a61b5096e4cdfe0c52a99a8","observation_id":"760bc7e5-e472-4085-b43d-a2c487a220c8","resolution":{"observed_at":"2026-07-04T11:49:50.790654Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.13306","snapshot_observed_at":"2026-08-01T05:50:29.618737Z","title":"OSUM: Advancing open speech understand- ing models with limited resources in academia,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22100","last_updated":"2026-07-24T08:52:44Z","snapshot_observed_at":"2026-08-13T04:53:52.000451Z","submitted_at":"2026-07-24T08:52:44Z","title":"MEUSLI: a Multilingual Projector for LLM-based ASR and Beyond","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T05:50:29.618737Z"},"links":{"cited_paper":"/paper/2501.13306","citing_paper":"/paper/2607.22100"},"observation_digest":"sha256:4f12c1038a13e2a546ec5844cd85d7dd931891efb8cf785591437093982c7aca","observation_id":"459daaf3-3a19-4091-a8cc-2608f1eb74b1","resolution":{"observed_at":"2026-08-01T05:50:29.618737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.13306/citation-record","integrity":"/paper/2501.13306/integrity","json":"/paper/2501.13306/citation-record.json","paper":"/paper/2501.13306"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.249877Z","title":"Data Products , 2024","venue":null,"work_id":"7e4b687f-bc37-41a9-b5b0-284609ce6c0b","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.641410Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:9838337f05405cba535c073c763eac3600dc2e3cdbb935ba6e9e6c66a13e3884","observation_id":"c4845ecc-ab65-4973-bb13-35ee447b6f80","resolution":{"observed_at":"2026-08-10T16:21:51.254183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04051","last_updated":"2024-07-11T02:08:35Z","snapshot_observed_at":"2026-08-14T17:40:31.626392Z","submitted_at":"2024-07-04T16:49:02Z","title":"FunAudioLLM: Voice Understanding and Generation Foundation Models for Natural Interaction Between Humans and LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04051","snapshot_observed_at":"2026-08-10T16:21:50.645939Z","title":"FunaudioLLM : Voice understanding and generation foundation models for natural interaction between humans and LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.645939Z"},"links":{"cited_paper":"/paper/2407.04051","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:720811d7dd8f52a385761317ee30709d443515a3b46e8b203a54d05f91709bad","observation_id":"0a0579ea-5f4d-4fa8-a1e7-9c9f4ee76847","resolution":{"observed_at":"2026-08-10T16:21:50.645939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-10T16:21:50.650773Z","title":"Qwen technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.650773Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:4f53219f004b3fa7743a8f9d12f141ffe48c9ddce9bcb4d248465b282d88d65a","observation_id":"26405e1f-01c6-4177-9c8e-34b8d827b0a4","resolution":{"observed_at":"2026-08-10T16:21:50.650773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.237562Z","title":"AISHELL-1 : An open-source mandarin speech corpus and a speech recognition baseline","venue":null,"work_id":"0f888a07-4cef-42ff-9682-1aa79db5509b","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.655249Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:bfa8115ecf48c459c735ec9476e725238654da6cfab9b90432b5a072c1766fe9","observation_id":"1d2f7d2d-bf49-4319-9d59-0b74ce33614c","resolution":{"observed_at":"2026-08-10T16:21:51.241788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.225589Z","title":"IEMOCAP : Interactive emotional dyadic motion capture database","venue":null,"work_id":"7ae0075c-6600-426d-a73d-7af303b115fb","year":2008},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.660136Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:b2815a35e2c71ffda11903099b96a49314237d4672c99efee0f2822b46d0d1fa","observation_id":"46728167-1fc3-45b8-b624-2198cb239ab6","resolution":{"observed_at":"2026-08-10T16:21:51.229299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.213771Z","title":"MSP-IMPROV : An acted corpus of dyadic interactions to study emotion perception","venue":null,"work_id":"f7d06017-e4ae-42dd-97f1-f5df75210cd6","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.664124Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:3a53bf5474b7b8162815c1543e8f67ff04e0e14e7b180d07405f6e6e1db1bbf1","observation_id":"52440562-690e-4e16-84cb-b880bc7a35ee","resolution":{"observed_at":"2026-08-10T16:21:51.217573Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-10T16:21:50.668642Z","title":"Qwen-Audio : Advancing universal audio understanding via unified large-scale audio-language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.668642Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:650b132e0df6b0c7c1c6f97751fd7b7539f5354e77df0fe44bececad8011897f","observation_id":"15110fa2-db69-4e3d-a69b-0a745922a32d","resolution":{"observed_at":"2026-08-10T16:21:50.668642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-10T16:21:50.672430Z","title":"Qwen2-Audio technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.672430Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:6efc5362d56b23d6f2dbf627b04a6adc4106db4ae3fc9d1eaa519f76c35ec993","observation_id":"953d4eb8-70f3-4527-b300-24d2dec5e61f","resolution":{"observed_at":"2026-08-10T16:21:50.672430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.202245Z","title":"Data products, 2024","venue":null,"work_id":"17423639-be07-4a44-9542-5af13e4d23a0","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.676243Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:a32e028822c146c0d364668e7c870c5e0814675dd0f793eefdb451affaa4adf1","observation_id":"2b3e2324-61fc-46a6-8ea9-9c3b754f2a8b","resolution":{"observed_at":"2026-08-10T16:21:51.205841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.191589Z","title":"Data products, 2024","venue":null,"work_id":"403df258-27d9-4462-90b3-1e0b86e937b0","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.679557Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:6363c74ba36e135e62e3b335de2e997e18ba51eb1e19cdaecb01bbeba9097d65","observation_id":"0013ddfe-9b79-4433-88e9-fedde57b0f45","resolution":{"observed_at":"2026-08-10T16:21:51.194923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1808.10583","last_updated":"2018-09-13T02:45:27Z","snapshot_observed_at":"2026-08-14T18:34:46.662561Z","submitted_at":"2018-08-31T03:11:08Z","title":"AISHELL-2: Transforming Mandarin ASR Research Into Industrial Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1808.10583","snapshot_observed_at":"2026-08-10T16:21:50.682871Z","title":"AISHELL-2 : Transforming Mandarin ASR research into industrial scale","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.682871Z"},"links":{"cited_paper":"/paper/1808.10583","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:a5554bcd0dea959b5fb771933b9d492b37a9abc00e24d225a9e7db2a9fbc7e60","observation_id":"ef476ca6-1f6f-4dad-8424-4384230db56e","resolution":{"observed_at":"2026-08-10T16:21:50.682871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.180883Z","title":"Gemmeke, Daniel P","venue":null,"work_id":"bf5f96ab-ab6d-4e85-8628-4873ae7dbf2c","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.688136Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:de25cd19a259fe1a43f26353ecac269d9934350cb3f475fe9b03361f1c9c5e21","observation_id":"02eea089-deb6-4bff-ab24-2e277fa8f5e9","resolution":{"observed_at":"2026-08-10T16:21:51.184074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.170190Z","title":"Vocalsound: A dataset for improving human vocal sounds recognition","venue":null,"work_id":"de60ac5e-9cae-471a-b296-9c60af4f317a","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.692800Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:08320f3557c755c6186e9514ff854a68345a7e8e8146583b19e927b3bbb13147","observation_id":"b763e6cd-8fb7-45a2-ac99-ef753df47ed6","resolution":{"observed_at":"2026-08-10T16:21:51.173901Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.160109Z","title":"LoRA : Low -rank adaptation of large language models","venue":null,"work_id":"f4015ca5-ef89-4fc8-a04d-8dcc538500e5","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.697178Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:33dd0b3ff421d4ddb877bcc8d6a68aa1bf9c583fd774697be27bb3346cae8a88","observation_id":"f7be6a17-c051-4edf-a4c8-4012dc488c1a","resolution":{"observed_at":"2026-08-10T16:21:51.163633Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.149627Z","title":"Datasets, 2017","venue":null,"work_id":"2b9f3c6a-3283-41fb-8744-26f4fbeac048","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.701458Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:cb7a25574fc63345ff52a832479383f2739aaed81365ce44de8be834a7172161","observation_id":"d2030894-3781-4c39-91b1-699135cd2541","resolution":{"observed_at":"2026-08-10T16:21:51.153105Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.138763Z","title":"Schuller, and Jianhua Tao","venue":null,"work_id":"e74e919d-66f2-46c5-87ce-ed492b9ddb1c","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.705820Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:df6d22fa40781cafd2fb126d9cde6cd8d527df9cc701342beb9abd7a5b2ec2bd","observation_id":"b228b375-8c03-4806-9680-93ae9f01d650","resolution":{"observed_at":"2026-08-10T16:21:51.142135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.127916Z","title":"Emotion2vec: Self-supervised pre-training for speech emotion representation","venue":null,"work_id":"fffd6da1-f113-4fd6-80c8-2c6897ebd418","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.709957Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:6661def6745bec4d3542b078b5f0c90c96faa29be6f574d907d7be080549d873","observation_id":"f66d6d84-5633-41ca-9be7-69b3c0021db3","resolution":{"observed_at":"2026-08-10T16:21:51.131662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.116237Z","title":"The MSP -conversation corpus","venue":null,"work_id":"fc6677a1-c986-40ce-be9e-e1170b92664f","year":2020},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.714460Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:bab277ec54b40b1d7392587dd2c5589d57af7f58c033827243aac71c577a1eea","observation_id":"fdfd17ed-1cf7-4c0e-be38-c59aaeadb688","resolution":{"observed_at":"2026-08-10T16:21:51.120340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.104166Z","title":"MAGICDATA mandarin Chinese read speech corpus, 2019","venue":null,"work_id":"16a846a7-921e-4a74-b02c-43bce049a01f","year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.718846Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:cae797937dae2c070340f605b70684a04d6c125bdba32ed5b7661a61786cf448","observation_id":"f9167552-5a2f-4ed4-9def-0c061fd591f3","resolution":{"observed_at":"2026-08-10T16:21:51.107988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.092321Z","title":"Librispeech: An ASR corpus based on public domain audio books","venue":null,"work_id":"d6ce973a-122a-4766-863c-5feb5f37bdb9","year":2015},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.724151Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:4533bcdf7a3f149ee20e56eb9b46a23c7814baaa77adccb90e3a8d6eeb45b757","observation_id":"a9ba5083-904c-4640-9995-ae09ad4aef9e","resolution":{"observed_at":"2026-08-10T16:21:51.096486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.079562Z","title":"Reproducing whisper-style training using an open-source toolkit and publicly available data","venue":null,"work_id":"4d96c3b4-63fa-4c2a-83f5-94d64077a80c","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.728388Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:c183dee055442547c5f9b583516464e11789fe4e75a030d0b1faf9bd63b78030","observation_id":"40c2663a-8e81-4e98-b77a-ac501ac14610","resolution":{"observed_at":"2026-08-10T16:21:51.084881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.069405Z","title":null,"venue":null,"work_id":"d0745753-371a-439a-bf2b-ed742826e9d1","year":2015},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.731940Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:9121672515e18322155645a402f4a28338a64aca32c339a9ca7e666097b464f4","observation_id":"a5df2899-512f-413b-933e-4b6cfa6d6f51","resolution":{"observed_at":"2026-08-10T16:21:51.072413Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.058256Z","title":"MELD : A multimodal multi-party dataset for emotion recognition in conversations","venue":null,"work_id":"e9eb1177-7fd9-4e93-9d20-448dd595548e","year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.735946Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:485dbc4a4d3ff22a834dc6c492ce64a4eef102f9031ddaa3a6eaa13bc706deb4","observation_id":"0c1bfbc9-4cb3-435b-9387-85f3af8b9463","resolution":{"observed_at":"2026-08-10T16:21:51.061704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.047668Z","title":"Robust speech recognition via large-scale weak supervision","venue":null,"work_id":"72976760-3f8c-44c6-becb-20432d427cc2","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.739872Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:7f77f5e5f576e1fe41b9d1eba74d8013f22445f92951e3e62b539404b5734afa","observation_id":"c4678dbb-125c-49d6-8e58-cd612f3a968a","resolution":{"observed_at":"2026-08-10T16:21:51.051351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.035672Z","title":"Nonspeech7k dataset: Classification and analysis of human non-speech sound","venue":null,"work_id":"5de055d4-fbd8-4e61-a7a7-b0facfcf9ffb","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.743915Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:f38c5b2800e9b4322d1dd21e0933f9bca2f30b0577b172232a5b61ac1a6343d5","observation_id":"495dd3a5-d28c-4dd6-a54d-f0e27ab3d062","resolution":{"observed_at":"2026-08-10T16:21:51.039921Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.05916","last_updated":"2020-07-12T05:38:57Z","snapshot_observed_at":"2026-08-13T22:09:27.930803Z","submitted_at":"2020-07-12T05:38:57Z","title":"The ASRU 2019 Mandarin-English Code-Switching Speech Recognition Challenge: Open Datasets, Tracks, Methods and Results","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.05916","snapshot_observed_at":"2026-08-10T16:21:50.748041Z","title":"The ASRU 2019 Mandarin - English code-switching speech recognition challenge: Open datasets, tracks, methods and results","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.748041Z"},"links":{"cited_paper":"/paper/2007.05916","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:e989672191864dc6e1b5c2889eb10780d20dbfb47a3cfef2af90970aa35b697e","observation_id":"64e3ad61-25f5-4207-90f2-ffa3e92de62b","resolution":{"observed_at":"2026-08-10T16:21:50.748041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.022090Z","title":"Achieving timestamp prediction while recognizing with non-autoregressive end-to-end ASR model","venue":null,"work_id":"40694906-be51-42c2-a10c-380684264462","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.751915Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:7433ad1d8fd724a5049f7cf3a4b1466df0f555d464ecaa1847e7f9a5b0217703","observation_id":"e20f0866-7a74-4ce1-8c57-5d36fc155563","resolution":{"observed_at":"2026-08-10T16:21:51.026269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15622","last_updated":"2024-12-20T07:28:04Z","snapshot_observed_at":"2026-08-13T18:08:12.828161Z","submitted_at":"2024-12-20T07:28:04Z","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","version":1},"cited_work":{"arxiv_id":"2412.15622","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.15622","snapshot_observed_at":"2026-08-10T16:21:50.826977Z","title":"TouchASP: Elastic Automatic Speech Perception that Everyone Can Touch","venue":"eess.AS","work_id":"87dfe593-0db0-4e11-aca7-24a3479f58d4","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.755899Z"},"links":{"cited_paper":"/paper/2412.15622","citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:394426e53a93ccd9ec8ecbe556140fed8aa15de7b2a30b78a035c50172126a5c","observation_id":"afe57756-a2d3-470d-80f4-fd23a3dffd5b","resolution":{"observed_at":"2026-08-10T16:21:50.833353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:51.010976Z","title":"PandaGPT : One model to instruction-follow them all","venue":null,"work_id":"96bcb1d9-e852-440b-a873-ea42aaa6f101","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.760638Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:9d4e2ad5374c9e1ac3fcd58ed0dcebd437fc0082daa23a8c99abaa24719fd733","observation_id":"84ec1fc9-4e59-48ac-8f98-fc5b362f883e","resolution":{"observed_at":"2026-08-10T16:21:51.014485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.999115Z","title":"SALMONN : Towards generic hearing abilities for large language models","venue":null,"work_id":"f1be6610-7358-4cc1-8412-69ea2b933e4d","year":2024},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.764260Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:849513b9d2f7253f89d057321488178f8c237afbc6448429f6c4a995cd01df98","observation_id":"5fcabe55-126a-48c5-bc0c-739cc90a3f2d","resolution":{"observed_at":"2026-08-10T16:21:51.002925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.987891Z","title":"Kespeech: An open source speech dataset of Mandarin and its eight subdialects","venue":null,"work_id":"a4b5bd2b-8121-4472-a295-252c46c18dc7","year":2021},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.767988Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:70e6a51135ab2e491c9afb710e42cc5b7816d9455fda4d46c3b2641f57410f30","observation_id":"2f9dd831-0c14-453c-9f29-ec7c9ad4fe32","resolution":{"observed_at":"2026-08-10T16:21:50.992013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.976037Z","title":"Upadhyay, Woan-Shiuan Chien, Bo-Hao Su, Lucas Goncalves, Ya-Tse Wu, Ali N","venue":null,"work_id":"7cd8f3e0-f6f0-41cf-b2d2-2855ad84d677","year":2023},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.771903Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:441a376955343551e9d3e03bd372ac455913612cea4ea8343aa1081743e2ba99","observation_id":"5eef7f39-a106-4968-9c33-982a1ac3b2c2","resolution":{"observed_at":"2026-08-10T16:21:50.980216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.963922Z","title":"Attention is all you need","venue":null,"work_id":"56d35d8f-a3e5-4b4a-85cc-decadcc9a0d3","year":2017},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.775557Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:0ea3ef8c4fffec045f8731e1cc338df19001cb1e6381def2e00efcf507d3f2b0","observation_id":"52d13f61-9502-479a-936d-b5b890a9a7ae","resolution":{"observed_at":"2026-08-10T16:21:50.967826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.951780Z","title":"A large-scale Chinese short-text conversation dataset","venue":null,"work_id":"39163cb5-0043-49ba-8504-0f27c5a40a8c","year":2020},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.779084Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:5ea88563581616bb38526b8df5f6041403a91f483798b736d6e47f3df4c263d7","observation_id":"caff80b2-1551-4bc4-b5e7-265f7e582f6b","resolution":{"observed_at":"2026-08-10T16:21:50.955944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.939016Z","title":"WENETSPEECH : A 10000+ hours multi-domain Mandarin corpus for speech recognition","venue":null,"work_id":"aaa35b52-4127-45a4-9137-182805fe20ae","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.783589Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:c7b7b7f4b772c69386ee24c8b70b40918dc6156c4ff18f650c885be5ef57b96c","observation_id":"75834008-8b63-4873-963e-aa9b98735c35","resolution":{"observed_at":"2026-08-10T16:21:50.943291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.926518Z","title":"M 3 ED : Multi -modal multi-scene multi-label emotional dialogue database","venue":null,"work_id":"fd28aebe-16d7-4b50-bd12-7d797cc66aa4","year":2022},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.788155Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:efbc09cae5bf68b41a353432c4f83c92862e9a824c2a7d8f1f8c3e2e705cc134","observation_id":"98512b3b-063f-4236-be2d-d25226d42cf0","resolution":{"observed_at":"2026-08-10T16:21:50.930655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T16:21:50.912713Z","title":"Seen and unseen emotional style transfer for voice conversion with a new emotional speech dataset","venue":null,"work_id":"a59510a2-5786-40c4-b5f6-cae484b15cea","year":2021},"citing_paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-10T16:21:50.792032Z"},"links":{"citing_paper":"/paper/2501.13306"},"observation_digest":"sha256:7b56661fc33befd9e8b202cc911fc467bfc6e891cab10693e1bd2bf9d7917b04","observation_id":"26a6b988-ecf9-42c2-8381-48f36e2c8c64","resolution":{"observed_at":"2026-08-10T16:21:50.916362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.13306","last_updated":"2025-02-16T08:03:17Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-14T05:23:17.963963Z","submitted_at":"2025-01-23T01:27:46Z","title":"OSUM: Advancing Open Speech Understanding Models with Limited Resources in Academia"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":1,"verified_fuzzy":29},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 12 inbound Pith citation observations for arXiv:2501.13306."}