{"as_of":"2026-08-17T12:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:50b4470e84899944e035f613bf2b78b3ab003d93fd6528258a913cf9b2524929","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:23.320594Z","state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:26:22.389818Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T07:39:48.680240Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-08-07T11:26:22.389818Z","title":"Recent advancements in large language models (LLMs) have led to remarkable breakthroughs in speech LLMs and voice assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.389818Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:0034115ba8fce44efdd3dfd699b00f38314b8ca05ac4951a3b231843d5c6767e","observation_id":"24a79a7e-e7f3-46b6-8e29-f2eabedc2326","resolution":{"observed_at":"2026-08-07T11:26:22.389818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":"2506.02457","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sova-bench: Benchmarking the speech con- versation ability for llm-based voice assistant","venue":null,"work_id":"46008301-0f06-45f2-952f-b38d21b03566","year":2025},"citing_paper":{"arxiv_id":"2605.20266","last_updated":"2026-05-18T20:21:32Z","snapshot_observed_at":"2026-08-16T00:00:14.809961Z","submitted_at":"2026-05-18T20:21:32Z","title":"A Survey of Large Audio Language Models: Generalization, Trustworthiness, and Outlook","version":1},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:23.099479Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2605.20266"},"observation_digest":"sha256:8c9f173780f70c193a2effe3e1cbf7e5162c13b47551b7fd04171e85826bf9a3","observation_id":"77213481-1a76-4b50-ab3d-5c2d1777249c","resolution":{"observed_at":"2026-05-21T07:39:48.681853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.02457/citation-record","integrity":"/paper/2506.02457/integrity","json":"/paper/2506.02457/citation-record.json","paper":"/paper/2506.02457"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.02457","snapshot_observed_at":"2026-08-07T11:26:22.389818Z","title":"Recent advancements in large language models (LLMs) have led to remarkable breakthroughs in speech LLMs and voice assistants","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.389818Z"},"links":{"cited_paper":"/paper/2506.02457","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:0034115ba8fce44efdd3dfd699b00f38314b8ca05ac4951a3b231843d5c6767e","observation_id":"24a79a7e-e7f3-46b6-8e29-f2eabedc2326","resolution":{"observed_at":"2026-08-07T11:26:22.389818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.672051Z","title":"Speech LLM Speech LLM extends the understanding capability to speech flow, performing modality alignment between speech and text via an encoder with adaptors","venue":null,"work_id":"be030667-7682-4a52-ab25-3a54357b5e8e","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.451786Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:35a9679cc5429857656aca6c60180818bae442db01c74eafcfb2dea1f310ec19","observation_id":"6935127e-fa43-4f0d-b77e-26f4d1a3a331","resolution":{"observed_at":"2026-08-07T11:26:26.828263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.554859Z","title":null,"venue":null,"work_id":"40d028fb-0da7-4556-8190-d6fa0a17fd76","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.535835Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:b4c631858a1d3e511a0e4f88ad7fa099f47ccc2d1fcc8c79586b701c5a62c159","observation_id":"335ffbc3-ee3e-4578-a789-24643a2af936","resolution":{"observed_at":"2026-08-07T11:26:26.608938Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"1637.8469","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.775478Z","title":null,"venue":null,"work_id":"7539834e-9728-44f1-9575-428c65a03879","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.625987Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:771cb49b2519095c972b9e2544f4a5755f6a3f068737b25fc89f3fa1393494af","observation_id":"3007e468-ce9c-436f-8f31-d8c14bf5e103","resolution":{"observed_at":"2026-08-07T11:26:23.863004Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.395268Z","title":null,"venue":null,"work_id":"0b34e5c9-e528-4844-a1fc-b68fbb04ecff","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.693809Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:dff069d1ff604d6ae950a2754b2de0277b661cedd9428a759d97c730950c8577","observation_id":"741100b8-5df6-4d9e-8de8-ec39610b01af","resolution":{"observed_at":"2026-08-07T11:26:26.436653Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.230509Z","title":null,"venue":null,"work_id":"216a0ee5-d1e3-476a-a20d-9701448a08a3","year":null},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.770624Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:7f85a6e18edfed748b11bbcebf2ddbcb6cdd8caa59c0f9f718352cf5ae6c9fec","observation_id":"eed943fb-c74a-42e8-86bb-cac7be6255bb","resolution":{"observed_at":"2026-08-07T11:26:26.295669Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-08-15T14:02:47.366139Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-07T11:26:22.870915Z","title":"GPT-4o system card,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.870915Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:540b7b6c69929f1c4e8d71e84f87988e4b98f34e31448faee95290a83931dd75","observation_id":"04f99b43-0c47-40cd-b373-185506acfab5","resolution":{"observed_at":"2026-08-07T11:26:22.870915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16725","last_updated":"2024-11-05T02:24:18Z","snapshot_observed_at":"2026-08-16T13:22:50.721442Z","submitted_at":"2024-08-29T17:18:53Z","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16725","snapshot_observed_at":"2026-08-07T11:26:22.926836Z","title":"Mini-Omni: Language models can hear, talk while thinking in streaming,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:22.926836Z"},"links":{"cited_paper":"/paper/2408.16725","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:e8c31ba525203812de58959e7176f490c85c3219072ad5361ddb0a783b402dc8","observation_id":"b1ef39e5-7134-495b-aee8-c300409e6154","resolution":{"observed_at":"2026-08-07T11:26:22.926836Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-08-16T13:19:40.907971Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-07T11:26:23.008363Z","title":"LLaMA-Omni: Seamless speech interaction with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.008363Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:80e64f7081de2c35097a6da5a909259585692b16c45358367810e6e79e39ee51","observation_id":"3a38c437-6768-4ac8-b6b1-defd92aa8332","resolution":{"observed_at":"2026-08-07T11:26:23.008363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-07T11:26:23.050878Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.050878Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:1e3c85eeeb5dc186a4357f3b0ace61116e6a9296bbc0e7652782c67364824504","observation_id":"f821b026-4326-4f47-b148-bb854874ed7f","resolution":{"observed_at":"2026-08-07T11:26:23.050878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:26.110174Z","title":"Dynamic- SUPERB: Towards a dynamic, collaborative, and comprehensive instruction-tuning benchmark for speech,","venue":null,"work_id":"04713159-10a5-4961-93ac-51dfc03fff55","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.138843Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:79c5a9cc47c591fa4338c761233e7c06f89675f9d2e38203d096bbd962834dbd","observation_id":"657ac16c-0126-4684-af44-6e56729f7170","resolution":{"observed_at":"2026-08-07T11:26:26.163525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.05361","last_updated":"2025-06-09T16:36:12Z","snapshot_observed_at":"2026-08-16T13:01:51.403793Z","submitted_at":"2024-11-08T06:33:22Z","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.05361","snapshot_observed_at":"2026-08-07T11:26:23.197926Z","title":"Dynamic- SUPERB Phase-2: A collaboratively expanding benchmark for measuring the capabilities of spoken language models with 180 tasks,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.197926Z"},"links":{"cited_paper":"/paper/2411.05361","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:ccb73e60c57525a1bb3be8335432f024640a38a87d89e762921694d853d9021d","observation_id":"5aad3ced-54b1-4a76-8309-61beb49207e2","resolution":{"observed_at":"2026-08-07T11:26:23.197926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16020","last_updated":"2025-05-06T00:52:19Z","snapshot_observed_at":"2026-08-16T13:40:26.464210Z","submitted_at":"2024-06-23T05:40:26Z","title":"AudioBench: A Universal Benchmark for Audio Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16020","snapshot_observed_at":"2026-08-07T11:26:23.203188Z","title":"AudioBench: A universal benchmark for audio large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.203188Z"},"links":{"cited_paper":"/paper/2406.16020","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:59feb95138e6a655b6ca358a0ed37212c6896beee7df67957b6448ebfecf3f2d","observation_id":"0db694d5-841a-41cf-99e1-1c9efe1a8c85","resolution":{"observed_at":"2026-08-07T11:26:23.203188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07729","last_updated":"2024-07-26T06:30:47Z","snapshot_observed_at":"2026-08-16T14:19:18.351061Z","submitted_at":"2024-02-12T15:41:22Z","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07729","snapshot_observed_at":"2026-08-07T11:26:23.208362Z","title":"AIR-Bench: Benchmarking large audio- language models via generative comprehension,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.208362Z"},"links":{"cited_paper":"/paper/2402.07729","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:8182a54422080632db756ed2fab3a434d8bb45b85ff1c19a962b535ec861af76","observation_id":"1dae2d60-7300-4edf-ad27-3acb7069dff2","resolution":{"observed_at":"2026-08-07T11:26:23.208362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-08-13T13:23:17.901906Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-07T11:26:23.213004Z","title":"V oiceBench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.213004Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:d98d16e9c25d616952b380d481a3cc3fb3569bcf614befde27cc3877acce158a","observation_id":"277382e1-99a2-4fc9-ab1b-e2432272bb66","resolution":{"observed_at":"2026-08-07T11:26:23.213004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.988780Z","title":"SALMONN:Towards generic hearing abilities for large language models,","venue":null,"work_id":"a64edb40-dbba-48f3-a0cc-543679775d0e","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.217381Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:346dbf567a351d83421c99e827fa9b62cd867be55c708039320e12d1cd2b55e1","observation_id":"3208b10f-2a13-4b05-85bf-ef46656c5d91","resolution":{"observed_at":"2026-08-07T11:26:26.018205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.816990Z","title":"SpeechGPT: Empowering large language models with intrinsic cross-modal conversational abilities,","venue":null,"work_id":"efb3446f-1bfc-4556-bbdd-824c66ee2d2b","year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.222170Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:a68c0fc353ccd4e85963be81a5d73a17c0359a7ae7a2baed63a8250219492b59","observation_id":"c243da26-5a10-41da-9c47-b0fc12b019fb","resolution":{"observed_at":"2026-08-07T11:26:25.916923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-07T11:26:23.226286Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.226286Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:de96a9d42f97a626e5d4866a06d0fddbf30e3caf18094e7a2c3b86c77ad56e45","observation_id":"2f99f1ab-e977-4b39-b9f2-dc13e6840bf7","resolution":{"observed_at":"2026-08-07T11:26:23.226286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.701614Z","title":"SNAC: Multi- scale neural audio codec,","venue":null,"work_id":"28b5eb90-4bb6-4f96-b4fb-bd1ef92efff0","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.230921Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:2d13e9d9989a848ff3f82a4975233c59961cd1769e241dd0da3f4b689427d72b","observation_id":"d72b60f0-b1c0-4ad4-86ed-f944745a5b04","resolution":{"observed_at":"2026-08-07T11:26:25.756587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.234959Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.234959Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:4ffa3f6bc9bd4342722d087bb479965e63d46af2687b6a51b8c3c57a83b44464","observation_id":"78bf5a99-e0a4-4aac-ba3e-64c54fd2a8ab","resolution":{"observed_at":"2026-08-07T11:26:23.234959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.559866Z","title":"HiFi-GAN: Generative adversar- ial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"80bfa388-5d05-4e75-ad84-03d9e4a106a4","year":2020},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.239575Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:a3116f3f8eedea7e96f67531737642c02058fe074f466e4ea11acaf39d4a5c65","observation_id":"25d6d93a-a35b-43e3-a76c-698b211e4008","resolution":{"observed_at":"2026-08-07T11:26:25.610697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.369020Z","title":"Speech resynthesis from discrete disentangled self-supervised representations,","venue":null,"work_id":"65a43c59-4527-46af-a95d-69d954c59a65","year":2021},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.244104Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:9179819d0ce47dec2918160fd0395552d1929f8083c0f1cb272004698082f3a4","observation_id":"81babd28-831b-43ef-be8d-cef09381c961","resolution":{"observed_at":"2026-08-07T11:26:25.430775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:25.164843Z","title":"Westlake-Omni,","venue":null,"work_id":"0b3bec86-a6d0-4bc3-be74-df9898f0f636","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.248523Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f5133f1ac3f007aa7e26b70cd2a19a6afa4fd86ec84de69fa37800f7f43409f9","observation_id":"37d13974-9f50-4609-bf99-3a907d1da2c4","resolution":{"observed_at":"2026-08-07T11:26:25.265736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00774","last_updated":"2024-12-08T05:41:56Z","snapshot_observed_at":"2026-08-16T13:03:42.737566Z","submitted_at":"2024-11-01T17:59:51Z","title":"Freeze-Omni: A Smart and Low Latency Speech-to-speech Dialogue Model with Frozen LLM","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00774","snapshot_observed_at":"2026-08-07T11:26:23.253202Z","title":"Freeze- Omni: A smart and low latency speech-to-speech dialogue model with frozen llm,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.253202Z"},"links":{"cited_paper":"/paper/2411.00774","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:f4dbbd5ab4491e7e0cdb5607711d4a4be38b503ef86b7e2af7b1021cfd9583a3","observation_id":"fd4d79c4-b7f4-466a-bb6d-a80d9ec59e96","resolution":{"observed_at":"2026-08-07T11:26:23.253202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.969618Z","title":"Be- yond turn-based interfaces: Synchronous llms as full-duplex dia- logue agents,","venue":null,"work_id":"f2f46819-3e67-4bcf-ad9a-67f5aaa1d25f","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.257541Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:6f3814d22a670b39ce74e8f52a67c7374464ea50685d0cbbc6b857ed2e6e6e00","observation_id":"f725328a-7a76-49dc-a579-cb24b7ecc36c","resolution":{"observed_at":"2026-08-07T11:26:25.096745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17799","last_updated":"2025-01-03T06:15:58Z","snapshot_observed_at":"2026-08-16T13:06:39.141902Z","submitted_at":"2024-10-23T11:58:58Z","title":"OmniFlatten: An End-to-end GPT Model for Seamless Voice Conversation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17799","snapshot_observed_at":"2026-08-07T11:26:23.261871Z","title":"OmniFlatten: An end-to-end GPT model for seamless voice conversation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.261871Z"},"links":{"cited_paper":"/paper/2410.17799","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:ce9ec5e964106cc3c48514770683ee185199c8cf6c8da7e47ef6545513f46fff","observation_id":"101c2dd5-5392-419c-820a-30b3c3d19b08","resolution":{"observed_at":"2026-08-07T11:26:23.261871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.266830Z","title":"Baichuan-Omni-1.5 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.266830Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:3bc7dc0f590d5524b906684dbdaf9cbc4a5216922042c22adc39aebe48463297","observation_id":"b577a389-a753-4102-adbf-4c98624ca753","resolution":{"observed_at":"2026-08-07T11:26:23.266830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.807192Z","title":"TriviaQA: A large scale distantly supervised challenge dataset for reading comprehension,","venue":null,"work_id":"3a347501-a554-4349-92be-e1ae803d62df","year":2017},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.271445Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:dad5752b06423553f01903e0ad67339c5dee8d2583ee15fa0192c425da3b9ce5","observation_id":"f9735e47-eb7d-479c-a8ac-03dc3d0f8cf8","resolution":{"observed_at":"2026-08-07T11:26:24.898400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.276005Z","title":"Lib- rispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.276005Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:70d2beeb63760c909aa0b21ab7b563b034610ae52e98b2906486ad4c436506b8","observation_id":"2733f0ee-bbe7-44dd-b717-497f1ce9d094","resolution":{"observed_at":"2026-08-07T11:26:23.276005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.633120Z","title":"LibriSQA: A novel dataset and framework for spoken question answering with large language models,","venue":null,"work_id":"02957620-88da-4e6a-9150-ab0d27bd6dbd","year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.280776Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:57be3d639f0f1551e63ab846ba9757e5a129c2495e8624f6f55ae2a2948ba549","observation_id":"ffa586ec-7787-4748-bc32-a3b4aa3794d3","resolution":{"observed_at":"2026-08-07T11:26:24.706381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.498269Z","title":"Spoken SQuAD: A study of mitigating the impact of speech recognition errors on listening comprehension,","venue":null,"work_id":"3f2b54c7-a558-42f5-ae18-026cd5094027","year":2018},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.285411Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:d6feb4b1f7e9a55c006dffb3f828d6cd995108097c67626d5e50aec6bcb854c2","observation_id":"4ad395e0-aac3-43e9-820e-feafba8e888d","resolution":{"observed_at":"2026-08-07T11:26:24.560476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.289433Z","title":"IEMOCAP: Interactive emotional dyadic motion capture database,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.289433Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:266f4e5b524614aabd6bfa610683e493e82e4f0a4d6dbb2d43cbdbde7211facd","observation_id":"6d061c8f-99cc-4c34-97b1-977e6766fafb","resolution":{"observed_at":"2026-08-07T11:26:23.289433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.237193Z","title":"Common V oice: A massively-multilingual speech corpus,","venue":null,"work_id":"cea79bc6-bd9e-4ed8-8ae0-6983fe939044","year":2020},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.294124Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:15e47a390b97f0dc81320c32df0d4204fccf4799e7387981949eb1466e672ea1","observation_id":"3f915456-202c-420d-b2bd-e2b094f68824","resolution":{"observed_at":"2026-08-07T11:26:24.374900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:24.035097Z","title":"Stanford alpaca: an instruction- following llama model (2023),","venue":null,"work_id":"1601a21f-7e25-4ec2-bbd7-bea8804dd428","year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.298170Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:2af86ee4a7e571dd6236ade92e9bccfa17da772c3f72dbe33b4cb0c7a9e9c7ef","observation_id":"16d4b01d-6eed-45ce-aa65-308b97d668b2","resolution":{"observed_at":"2026-08-07T11:26:24.140936Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.09305","last_updated":"2024-09-14T05:03:18Z","snapshot_observed_at":"2026-08-16T13:18:37.659352Z","submitted_at":"2024-09-14T05:03:18Z","title":"The T05 System for The VoiceMOS Challenge 2024: Transfer Learning from Deep Image Classifier to Naturalness MOS Prediction of High-Quality Synthetic Speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.09305","snapshot_observed_at":"2026-08-07T11:26:23.302596Z","title":"The t05 sys- tem for the voicemos challenge 2024: Transfer learning from deep image classifier to naturalness mos prediction of high-quality syn- thetic speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.302596Z"},"links":{"cited_paper":"/paper/2409.09305","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:c2203ae5bb2fff9f91f62b1714b1eee47f9de7ecbc0cb22299140160f6040bfb","observation_id":"3ca2bfd7-9187-4d2a-945d-c5e9210fa9d0","resolution":{"observed_at":"2026-08-07T11:26:23.302596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-07T11:26:23.307131Z","title":"Cosyvoice: A scalable multi- lingual zero-shot text-to-speech synthesizer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.307131Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:602924099a8ca20fd10885f6b0ab5af657a134f865f0900e217633b2a1832870","observation_id":"f82a96c8-3bec-43ff-899d-1ad5b1167973","resolution":{"observed_at":"2026-08-07T11:26:23.307131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:26:23.311743Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.311743Z"},"links":{"citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:605f3ef5b179825b2ebfc9003ad998de7cdaddd451cf917b42e23d9c462987b0","observation_id":"3141e860-23e4-4cdc-b2df-98ae5a89919b","resolution":{"observed_at":"2026-08-07T11:26:23.311743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11190","last_updated":"2024-11-05T02:27:57Z","snapshot_observed_at":"2026-08-16T13:09:22.197235Z","submitted_at":"2024-10-15T02:10:45Z","title":"Mini-Omni2: Towards Open-source GPT-4o with Vision, Speech and Duplex Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11190","snapshot_observed_at":"2026-08-07T11:26:23.315974Z","title":"Mini-Omni2: Towards open-source GPT- 4o with vision, speech and duplex capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.315974Z"},"links":{"cited_paper":"/paper/2410.11190","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:d4fd164bc53a87ffe482deae4098b59d28bb9e0f5554c67ea226f104dafcdfaf","observation_id":"4b7c8061-2bde-426b-8c21-ff56e53f4e71","resolution":{"observed_at":"2026-08-07T11:26:23.315974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02612","last_updated":"2024-12-03T17:41:24Z","snapshot_observed_at":"2026-08-12T12:50:46.995038Z","submitted_at":"2024-12-03T17:41:24Z","title":"GLM-4-Voice: Towards Intelligent and Human-Like End-to-End Spoken Chatbot","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02612","snapshot_observed_at":"2026-08-07T11:26:23.320594Z","title":"GLM-4-V oice: Towards intelligent and human-like end- to-end spoken chatbot,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:26:23.320594Z"},"links":{"cited_paper":"/paper/2412.02612","citing_paper":"/paper/2506.02457"},"observation_digest":"sha256:175ebc0c709997b1f12748bdbe7631005f93ced7d181a3f4a35f955d7e0ac787","observation_id":"8b15efbd-c7c1-496d-b92e-9470a0a2afa2","resolution":{"observed_at":"2026-08-07T11:26:23.320594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.02457","last_updated":"2025-06-03T05:21:51Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-14T21:56:28.538771Z","submitted_at":"2025-06-03T05:21:51Z","title":"SOVA-Bench: Benchmarking the Speech Conversation Ability for LLM-based Voice Assistant"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":24,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 2 inbound Pith citation observations for arXiv:2506.02457."}