{"as_of":"2026-08-10T06:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:21e3e39e4e68c434a713b69432d7e5989290bac422da56944fa601f05fffcee6","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":17,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":17,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T10:36:06.472390Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2505.15957","last_updated":"2026-04-26T17:05:53Z","snapshot_observed_at":"2026-07-06T21:28:06.120576Z","submitted_at":"2025-05-21T19:17:29Z","title":"Towards Holistic Evaluation of Large Audio-Language Models: A Comprehensive Survey","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T13:32:57.771753Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2505.15957"},"observation_digest":"sha256:a6f08d18c8dfea219cc36ad517da0970a893863d037e0479eeab3a85d9ec1fab","observation_id":"db7cb67f-4973-4c2f-9e03-74dc6668fbff","resolution":{"observed_at":"2026-05-22T13:34:53.242696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2507.16632","last_updated":"2025-08-27T16:42:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-22T14:23:55Z","title":"Step-Audio 2 Technical Report","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:50.900436Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2507.16632"},"observation_digest":"sha256:edad195311f3660c86faa8c41fbd2a4f8f4ef6bc2cff410ebe75b7ac696277ec","observation_id":"106fd36e-5574-4659-a943-269086e2d72c","resolution":{"observed_at":"2026-05-16T05:59:51.100939Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T10:36:06.472390Z","title":"Zeng, A.; Du, Z.; Liu, M.; Wang, K.; Jiang, S.; Zhao, L.; Dong, Y .; and Tang, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.03940","last_updated":"2025-09-04T07:03:46Z","snapshot_observed_at":"2026-08-08T20:51:35.281735Z","submitted_at":"2025-09-04T07:03:46Z","title":"VoxRole: A Comprehensive Benchmark for Evaluating Speech-Based Role-Playing Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T10:36:06.472390Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2509.03940"},"observation_digest":"sha256:97bc2feed076b71cabf9e1068c8b6b86c9edb8545478d0b33fc369bd36dbbe79","observation_id":"d17a3726-3a7b-4486-a695-6b0d4d77566b","resolution":{"observed_at":"2026-08-05T10:36:06.472390Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2509.26388","last_updated":"2026-05-01T17:28:37Z","snapshot_observed_at":"2026-07-06T22:31:15.349221Z","submitted_at":"2025-09-30T15:23:39Z","title":"Game-Time: Evaluating Temporal Dynamics in Spoken Language Models","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T11:51:43.561210Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2509.26388"},"observation_digest":"sha256:34d5651d53a4c619b429867fc8783353251b67110f43897b159cea1bb923fe43","observation_id":"ccd32e67-ce37-47b6-92ba-2e7f39352a61","resolution":{"observed_at":"2026-05-18T11:52:35.642915Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2510.09592","last_updated":"2026-05-10T15:21:35Z","snapshot_observed_at":"2026-07-06T22:32:22.953630Z","submitted_at":"2025-10-10T17:50:59Z","title":"Mind-Paced Speaking: A Dual-Brain Approach to Real-Time Reasoning in Spoken Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-18T07:43:23.913399Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2510.09592"},"observation_digest":"sha256:803002422ce0e16e3e30599330cb543b734d0b5a9536470f69874c1fb277d205","observation_id":"b60aa5e7-3069-4b7e-a59c-a4ba0bbc85f4","resolution":{"observed_at":"2026-05-18T07:46:03.542900Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-04T10:13:42.937202Z","title":"Qian Yang, Jin Xu, Wenrui Liu, Yunfei Chu, Ziyue Jiang, Xiaohuan Zhou, Yichong Leng, Yuanjun Lv, Zhou Zhao, Chang Zhou, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2510.11098","last_updated":"2026-07-06T08:06:32Z","snapshot_observed_at":"2026-08-08T04:21:21.183477Z","submitted_at":"2025-10-13T07:45:52Z","title":"VCB Bench: An Evaluation Benchmark for Audio-Grounded Large Language Model Conversational Agents","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T10:13:42.937202Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2510.11098"},"observation_digest":"sha256:54b989feae9048cadeaccf9b1ef4ccf381a8762cc31af81761fd81d80f7929eb","observation_id":"aba706f3-94e5-400c-b97d-419f63c5aace","resolution":{"observed_at":"2026-08-04T10:13:42.937202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2604.15804","last_updated":"2026-04-21T03:35:14Z","snapshot_observed_at":"2026-08-01T22:56:50.755050Z","submitted_at":"2026-04-17T08:05:46Z","title":"Qwen3.5-Omni Technical Report","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T08:11:22.402552Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2604.15804"},"observation_digest":"sha256:49e5d93cb597e669b104655fc301f97243a84a83a63ae9cc5eaaf05d97bdb2cc","observation_id":"c3c10ec2-4a90-4f64-8d47-f4c4f0029954","resolution":{"observed_at":"2026-05-10T08:12:26.504670Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.06765","last_updated":"2026-05-07T17:59:56Z","snapshot_observed_at":"2026-08-08T18:16:48.532228Z","submitted_at":"2026-05-07T17:59:56Z","title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-11T01:03:09.942984Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.06765"},"observation_digest":"sha256:ec3e4a1b257b151707f96d3e0a23393992433cda89fcb2e7fc63098b61e69df2","observation_id":"5be8038e-bcc9-4550-ba44-09d2b089ce63","resolution":{"observed_at":"2026-05-11T04:50:56.186750Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.09413","last_updated":"2026-05-10T08:28:58Z","snapshot_observed_at":"2026-07-06T23:21:30.897592Z","submitted_at":"2026-05-10T08:28:58Z","title":"Evaluating the Expressive Appropriateness of Speech in Rich Contexts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-12T01:56:11.473572Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.09413"},"observation_digest":"sha256:6a05b9317a1b68e2a226a2db802337ad63a7ff7300613203032f3aa9c1388f7a","observation_id":"cab538ae-c1ca-426a-bfce-4a2d31dfbdcf","resolution":{"observed_at":"2026-05-12T01:56:14.573707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-03T05:12:45.221230Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":181,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:2b85f1676cd9b888c5d782d63bd2e1187f2223f43e2329d5098dbcc5e2067073","observation_id":"47f6c4d8-2d33-41ab-8c32-a3db3e5085b2","resolution":{"observed_at":"2026-05-21T07:39:49.047504Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20755","last_updated":"2026-06-11T05:13:51Z","snapshot_observed_at":"2026-08-02T13:39:32.672112Z","submitted_at":"2026-05-20T05:54:08Z","title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-21T02:41:13.583493Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20755"},"observation_digest":"sha256:eca7352f1f5c69b4b93524b3122fb965c4432ea965bf90849d87d26e63ae6915","observation_id":"330ba7bf-eb71-4b78-873f-bddb8994d002","resolution":{"observed_at":"2026-05-21T02:43:55.060576Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.20755","last_updated":"2026-06-11T05:13:51Z","snapshot_observed_at":"2026-08-02T13:39:32.672112Z","submitted_at":"2026-05-20T05:54:08Z","title":"DuplexSLA: A Full-Duplex Spoken Language Model with Synchronized Speech, Language, and Action","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-30T17:32:58.848455Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.20755"},"observation_digest":"sha256:dd540da193f88739ae98f7723dab648621b42c64f12ce7ebe42813d797fc747b","observation_id":"d981c1f0-b2bf-43c5-9c63-56797b1abc01","resolution":{"observed_at":"2026-06-30T17:34:57.301296Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":129,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:f2127b39bca2b82be1a1cd1ad5532448ae6d9f4f155045bf48e2a27bea02daad","observation_id":"bdd1c086-e010-41c2-8fd5-4207109c55dd","resolution":{"observed_at":"2026-05-21T02:09:24.437109Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.01016","last_updated":"2026-05-31T05:13:32Z","snapshot_observed_at":"2026-07-06T23:41:39.172310Z","submitted_at":"2026-05-31T05:13:32Z","title":"PolySpeech-100: A Large-Scale Benchmark for Speech Understanding Across 100+ Languages and Dialects","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-06-28T17:44:07.669223Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.01016"},"observation_digest":"sha256:08ad00fa5108da8b6fd58fb7bb1a1df2bfc85ff624c95a518e51a7748bd80d27","observation_id":"4cad0548-0107-41e4-a95f-f154e1f5ac0e","resolution":{"observed_at":"2026-06-28T17:52:27.009937Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.05896","last_updated":"2026-06-22T18:29:42Z","snapshot_observed_at":"2026-07-06T23:45:51.910263Z","submitted_at":"2026-06-04T09:03:43Z","title":"Resonant Minds: Closed-Loop Social Avatars with Theory of Mind","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T02:04:39.753443Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.05896"},"observation_digest":"sha256:e5a10ede12cae824cd4935b82dcdd9d07223268751f9a88e179732ec7b6b3413","observation_id":"b94e9da7-36be-463a-aa74-4ad1a086ea3d","resolution":{"observed_at":"2026-07-02T12:36:56.235234Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":"2502.17810","doi":"10.48550/arxiv.2502.17810","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chih-Kai Yang, Yu-Kuan Fu, Chen-An Li, Yi-Cheng Lin, Yu-Xiang Lin, Wei-Chih Chen, Ho Lam Chung, Chun-Yi Kuan, Wei-Ping Huang, Ke-Han Lu, and 1 others","venue":"ArXiv.org","work_id":"84beecc1-d27b-42f6-bb23-0db5fb15a00d","year":2025},"citing_paper":{"arxiv_id":"2606.07547","last_updated":"2026-05-04T17:54:41Z","snapshot_observed_at":"2026-08-07T09:08:57.491441Z","submitted_at":"2026-05-04T17:54:41Z","title":"Liberating LLM Capabilities in Full-Duplex Speech Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-01T00:13:14.701980Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2606.07547"},"observation_digest":"sha256:f3f545568eaff700cdb5f083e374e9f8673dc83546a04b8d34730186500dfa0f","observation_id":"30a4e08a-3965-4cca-931a-e968a2deb774","resolution":{"observed_at":"2026-07-01T00:15:09.030108Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17810","snapshot_observed_at":"2026-08-01T11:20:18.079021Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19932","last_updated":"2026-08-07T08:39:15Z","snapshot_observed_at":"2026-08-10T06:09:36.080821Z","submitted_at":"2026-07-22T09:04:55Z","title":"Efficient Chain-of-Modality Reasoning via Progressive Compression for Spoken Language Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-01T11:20:18.079021Z"},"links":{"cited_paper":"/paper/2502.17810","citing_paper":"/paper/2607.19932"},"observation_digest":"sha256:6d997659898cac4dae816ccd99a5e7fffcecea771992751c493be078d317f4d5","observation_id":"533cc13a-9506-4390-bb4e-c9a1d0468fef","resolution":{"observed_at":"2026-08-01T11:20:18.079021Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.17810/citation-record","integrity":"/paper/2502.17810/integrity","json":"/paper/2502.17810/citation-record.json","paper":"/paper/2502.17810"},"outbound":[],"paper":{"arxiv_id":"2502.17810","last_updated":"2025-08-10T12:34:42Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T17:51:23.956942Z","submitted_at":"2025-02-25T03:31:48Z","title":"URO-Bench: Towards Comprehensive Evaluation for End-to-End Spoken Dialogue Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 17 inbound Pith citation observations for arXiv:2502.17810."}