{"as_of":"2026-08-10T22:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ae653ee575569b3c00981e1db212dce02ff31b29d3212410ffc5f11337a10981","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:21:35.837431Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.08603/citation-record","integrity":"/paper/2507.08603/integrity","json":"/paper/2507.08603/citation-record.json","paper":"/paper/2507.08603"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.177367Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.177367Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:69f7acf1f2f79dc8e08287690268ef48492381e989db6706e6ff073e680fab67","observation_id":"b5d0901a-db98-4b33-a3cd-88193acda9cb","resolution":{"observed_at":"2026-08-06T18:21:31.177367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.223429Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.223429Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:db7020376129fef7fe69cbe0e04582c38fa80d32ab117d0be36835ee74ca2b89","observation_id":"bc1bf232-f64e-405c-892b-056f7b174a72","resolution":{"observed_at":"2026-08-06T18:21:31.223429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.250691Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.250691Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:b236de10823bdc9ad228e01e26d47d03f2bdb45271514c07511ee591b8f9e34c","observation_id":"77c5940c-87ad-49fb-b273-f7ab831b9f00","resolution":{"observed_at":"2026-08-06T18:21:31.250691Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.318511Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.318511Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:c48b2e76970fc8c43f9372e0c1c50770c5d166aa958ae35ab161b753ad113173","observation_id":"ce7f2e7b-ffb6-4eeb-ad87-43cef825ae54","resolution":{"observed_at":"2026-08-06T18:21:31.318511Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.391770Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.391770Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:36563a14dd1c0a7a2a2933c4a57121ccf19f98ac1987fb74e53ed34203a4aa07","observation_id":"e21ffa31-1134-4e9f-8417-fbd995b4fbc1","resolution":{"observed_at":"2026-08-06T18:21:31.391770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:38.438629Z","title":null,"venue":null,"work_id":"66ada214-2e8e-4e5c-9665-f8a786dd7d8f","year":null},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.487458Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:679f0628099adea4509a7d65c07d5c7957e2c2fdb4f82f253a438460e0cf4274","observation_id":"38c93169-86a2-4019-8674-0aaeb60de472","resolution":{"observed_at":"2026-08-06T18:21:38.489249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-10T14:07:02.234322Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-06T18:21:31.554203Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.554203Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:2b1080c62c49c942d4feaafa0597223661eb67784aa17b7a11574fd197afabc2","observation_id":"fa82fe1f-0030-4eda-b179-682d5017b10a","resolution":{"observed_at":"2026-08-06T18:21:31.554203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:38.282447Z","title":"Ardila, M","venue":null,"work_id":"7c32b58c-b4ea-4666-9b82-a2fe036e1bfe","year":2020},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.621200Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:e56d97e35cd3f37e80d9d5575a84429e8f9e13b6778e6efb5d0475c64d7be619","observation_id":"cc5b163b-83cf-42f6-8f4e-2109fd75fbff","resolution":{"observed_at":"2026-08-06T18:21:38.348451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12397","last_updated":"2024-06-18T08:38:59Z","snapshot_observed_at":"2026-08-09T19:14:22.161945Z","submitted_at":"2024-06-18T08:38:59Z","title":"Unveiling the Flaws: Exploring Imperfections in Synthetic Data and Mitigation Strategies for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12397","snapshot_observed_at":"2026-08-06T18:21:31.668305Z","title":"Unveiling the flaws: Exploring imperfections in synthetic data and mitigation strategies for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.668305Z"},"links":{"cited_paper":"/paper/2406.12397","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:99549fc80c8c3e0ee38d62fffed48756fdad42b7eeae7f0809ed407c98c8efb2","observation_id":"ac0ac4bd-f31d-4580-9285-1517dea39f23","resolution":{"observed_at":"2026-08-06T18:21:31.668305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-06T18:21:31.719560Z","title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.719560Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:352426642149f9b3f8340d500b400fd7fa7d52eac0d84176e2d0098f6108268e","observation_id":"993a590b-8453-4348-a822-eba41b16fcd7","resolution":{"observed_at":"2026-08-06T18:21:31.719560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-06T18:21:31.773559Z","title":"Qwen2-audio technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.773559Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:4a4f4010104c6d127ad7fb20e471517cb4293a27f885ba97660fbc212519dc12","observation_id":"25c746c5-d803-4157-82c2-ba454edf0306","resolution":{"observed_at":"2026-08-06T18:21:31.773559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08295","last_updated":"2025-03-24T21:06:53Z","snapshot_observed_at":"2026-07-06T18:13:55.543513Z","submitted_at":"2024-05-14T03:33:31Z","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08295","snapshot_observed_at":"2026-08-06T18:21:31.841304Z","title":"Speechverse: A large-scale generalizable audio language model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.841304Z"},"links":{"cited_paper":"/paper/2405.08295","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:7b9c41d8f464531357f44c223280636aaa9c4470838776d7dd196a5188eb31f4","observation_id":"fdb3b362-ebbd-455e-8dd6-55c2a7f4ad4f","resolution":{"observed_at":"2026-08-06T18:21:31.841304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:31.978385Z","title":"Liu, Ana Marasovi \\'c , Noah A","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:31.978385Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:070db3b7bc1bfe9ccb9c5f161b2a42c513e48991d94573e0795ba1deb0b9bac5","observation_id":"cb6540b1-98d2-45b1-b200-777f8d9bbe48","resolution":{"observed_at":"2026-08-06T18:21:31.978385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17146","last_updated":"2024-12-05T14:28:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-25T17:59:51Z","title":"Molmo and PixMo: Open Weights and Open Data for State-of-the-Art Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.17146","snapshot_observed_at":"2026-08-06T18:21:32.038607Z","title":"Smith, Hannaneh Hajishirzi, Ross Girshick, Ali Farhadi, and Aniruddha Kembhavi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.038607Z"},"links":{"cited_paper":"/paper/2409.17146","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:cec481b843d6cacd6d255db9069b95649aacbb01686c30eeb1f3405735eebcc5","observation_id":"4ed0a6a4-30fb-4de7-9769-f7ab26f67912","resolution":{"observed_at":"2026-08-06T18:21:32.038607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1903.00161","last_updated":"2019-04-16T21:22:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-03-01T05:32:01Z","title":"DROP: A Reading Comprehension Benchmark Requiring Discrete Reasoning Over Paragraphs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1903.00161","snapshot_observed_at":"2026-08-06T18:21:32.105451Z","title":"Drop: A reading comprehension benchmark requiring discrete reasoning over paragraphs, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.105451Z"},"links":{"cited_paper":"/paper/1903.00161","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:8cfc168ea06d42e5c3d49634711c12c1752de6748fef74d8e4a41e5207f13355","observation_id":"20ef30d2-fe62-40c5-8225-4279cb131f48","resolution":{"observed_at":"2026-08-06T18:21:32.105451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T18:21:32.173897Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.173897Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:efe4ed95de4cb2bd8898fec37ed8768cb809e65588b4be121ce02db42b8bc104","observation_id":"4c73894b-351a-4dc8-8aba-8e8f2af38bdc","resolution":{"observed_at":"2026-08-06T18:21:32.173897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.06666","last_updated":"2025-03-01T12:59:49Z","snapshot_observed_at":"2026-07-06T19:13:20.458958Z","submitted_at":"2024-09-10T17:34:34Z","title":"LLaMA-Omni: Seamless Speech Interaction with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.06666","snapshot_observed_at":"2026-08-06T18:21:32.233590Z","title":"Llama-omni: Seamless speech interaction with large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.233590Z"},"links":{"cited_paper":"/paper/2409.06666","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:74da8dd7f8b5e4e002feaa83f00328a20580ef8a88bb8dfae793933444a6bb47","observation_id":"60989bf5-f0e4-4c5f-ae7d-85c95a1d3c30","resolution":{"observed_at":"2026-08-06T18:21:32.233590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06753","last_updated":"2024-04-12T18:55:22Z","snapshot_observed_at":"2026-08-02T08:58:25.969928Z","submitted_at":"2023-11-12T06:56:14Z","title":"AudioChatLlama: Towards General-Purpose Speech Abilities for LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06753","snapshot_observed_at":"2026-08-06T18:21:32.287450Z","title":"Audiochatllama: Towards general-purpose speech abilities for llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.287450Z"},"links":{"cited_paper":"/paper/2311.06753","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:ec909b3cb70f0709f5aa5d4bec998123cd959e8269e1ae7b40fddf80e106c2f5","observation_id":"a14a1d4d-618f-479c-a7b3-02a5eb28c19f","resolution":{"observed_at":"2026-08-06T18:21:32.287450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14405","last_updated":"2023-12-10T22:50:21Z","snapshot_observed_at":"2026-07-06T16:23:30.533693Z","submitted_at":"2023-09-25T17:59:05Z","title":"Joint Audio and Speech Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14405","snapshot_observed_at":"2026-08-06T18:21:32.364376Z","title":"Liu, Hongyin Luo, Leonid Karlinsky, and James Glass","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.364376Z"},"links":{"cited_paper":"/paper/2309.14405","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:079b59cca39591c15187eb20000a7f5cb4e2a98b0b709d892a807a74fa94bd6a","observation_id":"9e97d071-a3d8-463d-81c2-56f79a24215b","resolution":{"observed_at":"2026-08-06T18:21:32.364376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:38.112354Z","title":"Gigaspeech: An evolving, multi-domain asr corpus with 10,000 hours of transcribed audio","venue":null,"work_id":"4aabaaf8-f553-4573-8861-985cb50d9304","year":2021},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.479371Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:80affc3d5a80b0798874b9a28e6ff0329178d41d66203af49bd2bb984588bf8a","observation_id":"bb53a03d-114b-4614-9b8e-9931f5de6c72","resolution":{"observed_at":"2026-08-06T18:21:38.173788Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:37.969436Z","title":"Text normalization challenge - english language","venue":null,"work_id":"e4b4c6c7-fa1e-4129-99c3-8ee3fa08f344","year":2017},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.614199Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:7af1b2c4015c30395d1edc79498a190afac524d22e9a655e311de53767aacdfa","observation_id":"a157b862-9bf2-4add-9f63-3e33abc2544a","resolution":{"observed_at":"2026-08-06T18:21:38.053014Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-08-07T07:43:16.294957Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-06T18:21:32.737282Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.737282Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:f6ac83398ce9b35251ce93b5fe2334ccbf66288076e3f27347d1fb32266032d1","observation_id":"eb22f036-411a-4ce0-9494-0aab2701595b","resolution":{"observed_at":"2026-08-06T18:21:32.737282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.07040","last_updated":"2017-12-19T16:48:05Z","snapshot_observed_at":"2026-07-31T01:20:29.856066Z","submitted_at":"2017-12-19T16:48:05Z","title":"The NarrativeQA Reading Comprehension Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.07040","snapshot_observed_at":"2026-08-06T18:21:32.893352Z","title":"The narrativeqa reading comprehension challenge, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.893352Z"},"links":{"cited_paper":"/paper/1712.07040","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:91c0dbb7e34d0e3509dcbf00f9f15594215db2b8b9d9fbbd50e2b17b45cea708","observation_id":"fd44c632-00c2-4a58-a534-21b4f1240a75","resolution":{"observed_at":"2026-08-06T18:21:32.893352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:37.828465Z","title":"Parler-tts","venue":null,"work_id":"5b9579ff-7419-48e0-8731-caf7e0694770","year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:32.989747Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:5f1561bb398330692384f186d6f169f6b5c71fb17787a518327fbd89e57b322a","observation_id":"efdd24cb-ffe3-494c-9669-1b144cfcf5ea","resolution":{"observed_at":"2026-08-06T18:21:37.899116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:33.101643Z","title":"Building and better understanding vision-language models: insights and future directions., 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.101643Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:b6acf945c9ed858ec581cda2c3e0b6ec0aa859be379a7bd7b6ac7863d2a57608","observation_id":"074156c3-d855-4e55-9ee5-e4aadce19174","resolution":{"observed_at":"2026-08-06T18:21:33.101643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15687","last_updated":"2023-10-19T13:23:28Z","snapshot_observed_at":"2026-08-10T13:43:11.982910Z","submitted_at":"2023-06-23T16:23:24Z","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15687","snapshot_observed_at":"2026-08-06T18:21:33.236994Z","title":"Voicebox: Text-guided multilingual universal speech generation at scale, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.236994Z"},"links":{"cited_paper":"/paper/2306.15687","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:9e2371483ce626c1322537b9ae24c5691c148bc4fa33307f9048b827440ca5c0","observation_id":"f142280b-7379-477a-bfee-909d5bfdb87c","resolution":{"observed_at":"2026-08-06T18:21:33.236994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:37.643570Z","title":"ROUGE : A package for automatic evaluation of summaries","venue":null,"work_id":"55443676-c68f-4027-83e6-ec2aa07cc795","year":2004},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.363752Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:ecb24400a85855c176681a43b006fad7fdda3e63b8181faea350796642011ef1","observation_id":"dde31491-b392-4225-9e63-5867ddcfad78","resolution":{"observed_at":"2026-08-06T18:21:37.743223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.05852","last_updated":"2019-12-15T17:32:45Z","snapshot_observed_at":"2026-08-06T12:08:10.170652Z","submitted_at":"2019-08-16T05:36:04Z","title":"Reasoning Over Paragraph Effects in Situations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1908.05852","snapshot_observed_at":"2026-08-06T18:21:33.466915Z","title":"Reasoning over paragraph effects in situations, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.466915Z"},"links":{"cited_paper":"/paper/1908.05852","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:136960e0fe861bdbf81047e48796aa0a42ca95458c7e731a5025c81f1435116f","observation_id":"b7deae69-fc8c-4ffe-a4bb-3700e3ad91ef","resolution":{"observed_at":"2026-08-06T18:21:33.466915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1405.0312","last_updated":"2015-02-21T01:48:49Z","snapshot_observed_at":"2026-07-06T03:42:37.507458Z","submitted_at":"2014-05-01T21:43:32Z","title":"Microsoft COCO: Common Objects in Context","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1405.0312","snapshot_observed_at":"2026-08-06T18:21:33.619224Z","title":"Lawrence Zitnick, and Piotr Dollár","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.619224Z"},"links":{"cited_paper":"/paper/1405.0312","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:d8a75e62ed5aa5645b6ba12c6505619982d8e35cf7f62e1533e8fdaa645bd1c8","observation_id":"5825da0a-f631-43d5-9c7f-9d7c24fdb946","resolution":{"observed_at":"2026-08-06T18:21:33.619224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.01144","last_updated":"2021-06-02T13:30:43Z","snapshot_observed_at":"2026-08-10T17:44:54.573997Z","submitted_at":"2021-06-02T13:30:43Z","title":"Towards Emotional Support Dialog Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.01144","snapshot_observed_at":"2026-08-06T18:21:33.739895Z","title":"Towards emotional support dialog systems, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.739895Z"},"links":{"cited_paper":"/paper/2106.01144","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:684af4e38b3ddc9ba0d9b4e74f8ec1ede70a96b8b1164968270431f5642b67d0","observation_id":"1d52b23c-4798-4a72-a484-890309dfc41e","resolution":{"observed_at":"2026-08-06T18:21:33.739895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01912","last_updated":"2024-02-02T21:29:34Z","snapshot_observed_at":"2026-08-10T22:40:48.420397Z","submitted_at":"2024-02-02T21:29:34Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01912","snapshot_observed_at":"2026-08-06T18:21:33.855401Z","title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.855401Z"},"links":{"cited_paper":"/paper/2402.01912","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:d96f5ccd577e6b5a82c424146f266c5202ff097417babe39cbab863b9a71a550","observation_id":"cdb68026-84dd-4f38-be9c-6b9913761e75","resolution":{"observed_at":"2026-08-06T18:21:33.855401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07316","last_updated":"2023-03-19T13:37:01Z","snapshot_observed_at":"2026-07-06T14:04:58.943383Z","submitted_at":"2022-10-13T19:42:08Z","title":"MTEB: Massive Text Embedding Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.07316","snapshot_observed_at":"2026-08-06T18:21:33.899270Z","title":"Mteb: Massive text embedding benchmark, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.899270Z"},"links":{"cited_paper":"/paper/2210.07316","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:25fa13fa8a9276fe60310acf7b49cc2e0c6cede9a2ab38bbe63c8667969303ac","observation_id":"91141d3e-b65c-4013-b8b3-923aa3136a45","resolution":{"observed_at":"2026-08-06T18:21:33.899270Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T18:21:33.970105Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:33.970105Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:a5d18c62eac676536fb96776f152ed9372a0dc8c856db1f8055673ab53f8e54a","observation_id":"10067679-e26e-48d0-8324-4ffc5123b096","resolution":{"observed_at":"2026-08-06T18:21:33.970105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:37.455945Z","title":"Text normalization","venue":null,"work_id":"738c108c-2ebb-4d6f-8d34-451c974d685d","year":2017},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.016172Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:17e76ea1bb9b7590ce8f140d4c196c1f63c9a6a2844f94418f9a5938174b92ed","observation_id":"5d55c66f-1eee-4960-a47c-ca7fd2154576","resolution":{"observed_at":"2026-08-06T18:21:37.555946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.03411","last_updated":"2020-12-19T09:18:21Z","snapshot_observed_at":"2026-07-06T10:21:14.277598Z","submitted_at":"2020-12-07T01:53:45Z","title":"MLS: A Large-Scale Multilingual Dataset for Speech Research","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.03411","snapshot_observed_at":"2026-08-06T18:21:34.080216Z","title":"Mls: A large-scale multilingual dataset for speech research","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.080216Z"},"links":{"cited_paper":"/paper/2012.03411","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:ae3c9eff3a6a1278748af209778046cbfc3852cd2e21b3f7287ceb26653b0271","observation_id":"877eaa05-1c5e-40e3-ad9d-252a336a1289","resolution":{"observed_at":"2026-08-06T18:21:34.080216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:37.191714Z","title":"Scaling speech technology to 1,000+ languages","venue":null,"work_id":"fc516065-cd95-42db-a0ac-7ee9b77391e7","year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.167671Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:056e137f3c53875eb5f7e7fa5716d047483d369faa70b90aa49e871839774381","observation_id":"fe30dfbd-5c11-419d-af98-e9754e4dfcad","resolution":{"observed_at":"2026-08-06T18:21:37.359210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-06T18:21:34.303507Z","title":"Robust speech recognition via large-scale weak supervision, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.303507Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:4597a45081db68ae25f9880ef93fb7457bcc50e2db8028880b4c98d46fbd6875","observation_id":"af316edf-f4b4-4b89-a2d7-6e53281a4ac4","resolution":{"observed_at":"2026-08-06T18:21:34.303507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.05250","last_updated":"2016-10-11T02:42:36Z","snapshot_observed_at":"2026-08-08T11:05:45.903773Z","submitted_at":"2016-06-16T16:36:00Z","title":"SQuAD: 100,000+ Questions for Machine Comprehension of Text","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.05250","snapshot_observed_at":"2026-08-06T18:21:34.401618Z","title":"Squad: 100,000+ questions for machine comprehension of text, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.401618Z"},"links":{"cited_paper":"/paper/1606.05250","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:bd4de044d7e1a3075ba69f02fbb677dd46075560ecc45cba7e8ae482faaf89cc","observation_id":"9cf829c4-689a-4601-89ef-96053eacbb5e","resolution":{"observed_at":"2026-08-06T18:21:34.401618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1806.03822","last_updated":"2018-06-11T06:10:11Z","snapshot_observed_at":"2026-08-10T03:45:00.871928Z","submitted_at":"2018-06-11T06:10:11Z","title":"Know What You Don't Know: Unanswerable Questions for SQuAD","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1806.03822","snapshot_observed_at":"2026-08-06T18:21:34.483569Z","title":"Know what you don't know: Unanswerable questions for squad, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.483569Z"},"links":{"cited_paper":"/paper/1806.03822","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:766c2954312e258f4fa350f07da8836c1ee20aa40899a3e61801f7eef2602477","observation_id":"0eb02243-b32f-44a4-810f-adbf1d80818d","resolution":{"observed_at":"2026-08-06T18:21:34.483569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05090","last_updated":"2024-04-07T22:15:13Z","snapshot_observed_at":"2026-08-10T16:50:33.843176Z","submitted_at":"2024-04-07T22:15:13Z","title":"How Bad is Training on Synthetic Data? A Statistical Analysis of Language Model Collapse","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05090","snapshot_observed_at":"2026-08-06T18:21:34.589888Z","title":"How bad is training on synthetic data? a statistical analysis of language model collapse, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.589888Z"},"links":{"cited_paper":"/paper/2404.05090","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:3544506c4d62505abb692668a21dda5bf0d938ef79402031a92000a1f7b841de","observation_id":"e9860435-a193-4322-89ae-96683773c6f5","resolution":{"observed_at":"2026-08-06T18:21:34.589888Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-05T06:18:44.577926Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-06T18:21:34.713583Z","title":"Llasm: Large language and speech model, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.713583Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:f3111080f02c1f8a77217a81afd05613026773b197a40cce5d6056ab8cea0c82","observation_id":"a5992823-3346-4bbb-9f37-b589bb962408","resolution":{"observed_at":"2026-08-06T18:21:34.713583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:36.944293Z","title":"Open automatic speech recognition leaderboard","venue":null,"work_id":"7b4ba865-c613-45d2-869a-b668ab5605c1","year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.846171Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:8dbe18e4519648b32ce01d0cea612902ce15b3665c1665ff1389c5050bd39a45","observation_id":"b9cf61c7-41c9-4312-8d77-6c0c96df84e1","resolution":{"observed_at":"2026-08-06T18:21:37.069001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.13289","last_updated":"2024-04-08T06:12:52Z","snapshot_observed_at":"2026-08-02T21:51:36.809095Z","submitted_at":"2023-10-20T05:41:57Z","title":"SALMONN: Towards Generic Hearing Abilities for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.13289","snapshot_observed_at":"2026-08-06T18:21:34.973274Z","title":"Salmonn: Towards generic hearing abilities for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.973274Z"},"links":{"cited_paper":"/paper/2310.13289","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:4ff14377d3fd25dbdd4ed4abacb43b1e39fdc821de6d291205ca79a64410fa1f","observation_id":"ef229f92-a1aa-43c2-bfb0-f3ca5b8f2429","resolution":{"observed_at":"2026-08-06T18:21:34.973274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.07990","last_updated":"2024-08-15T07:37:24Z","snapshot_observed_at":"2026-07-06T19:01:01.881616Z","submitted_at":"2024-08-15T07:37:24Z","title":"FuseChat: Knowledge Fusion of Chat Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.07990","snapshot_observed_at":"2026-08-06T18:21:35.074100Z","title":"Fusechat: Knowledge fusion of chat models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.074100Z"},"links":{"cited_paper":"/paper/2408.07990","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:db69e04f7a46e4a925f27cdcce3e03895447e5fa757879e26806c6d1e9dd46ca","observation_id":"e985b490-a722-4353-a956-d7f34b7dfbe2","resolution":{"observed_at":"2026-08-06T18:21:35.074100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02111","last_updated":"2023-01-05T15:37:15Z","snapshot_observed_at":"2026-08-07T10:11:17.796562Z","submitted_at":"2023-01-05T15:37:15Z","title":"Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02111","snapshot_observed_at":"2026-08-06T18:21:35.186108Z","title":"Neural codec language models are zero-shot text to speech synthesizers, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.186108Z"},"links":{"cited_paper":"/paper/2301.02111","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:595d921c8a92e0ff156fc00674732473f8c6194fdba668daf63f6f7ea0000796","observation_id":"996fba63-73dd-41c3-8edb-b77eb667202c","resolution":{"observed_at":"2026-08-06T18:21:35.186108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:36.754133Z","title":"Globe: A high-quality english corpus with global accents for zero-shot speaker adaptive text-to-speech, 2024","venue":null,"work_id":"ba95c3cc-a027-4d9e-9859-70e11fafff3e","year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.283724Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:adaee801642cc99bff09f1110f95cc4b6ab45e09b7d743f3d85e464f3193e456","observation_id":"312bf727-e800-4e66-b5bf-f64e8cde9eba","resolution":{"observed_at":"2026-08-06T18:21:36.851861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-06T18:21:35.396913Z","title":"Qwen2 technical report, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.396913Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:29b3b6476960d5adbbb322a928799cd2efb2402c07ce0d9e43d75e31c17095ad","observation_id":"d7a2212b-5c4e-4429-9e93-1f18d1d713eb","resolution":{"observed_at":"2026-08-06T18:21:35.396913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07729","last_updated":"2024-07-26T06:30:47Z","snapshot_observed_at":"2026-08-06T03:59:19.956442Z","submitted_at":"2024-02-12T15:41:22Z","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07729","snapshot_observed_at":"2026-08-06T18:21:35.520657Z","title":"Air-bench: Benchmarking large audio-language models via generative comprehension, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.520657Z"},"links":{"cited_paper":"/paper/2402.07729","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:94e346fc66e0e4d08102b80e9407a0a3943e76a378b05b1a900798e94b3c68b1","observation_id":"7415031a-9fde-4451-9531-a0ca47e2fe49","resolution":{"observed_at":"2026-08-06T18:21:35.520657Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18243","last_updated":"2024-03-27T04:20:18Z","snapshot_observed_at":"2026-08-09T20:11:53.005330Z","submitted_at":"2024-03-27T04:20:18Z","title":"Boosting Conversational Question Answering with Fine-Grained Retrieval-Augmentation and Self-Check","version":1},"cited_work":{"arxiv_id":"2403.18243","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.18243","snapshot_observed_at":"2026-08-06T18:21:36.019089Z","title":"Boosting Conversational Question Answering with Fine-Grained Retrieval-Augmentation and Self-Check","venue":"cs.AI","work_id":"12c28200-8cc5-4315-94b0-c19146824fbb","year":2024},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.616120Z"},"links":{"cited_paper":"/paper/2403.18243","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:86fb0924af4d610901f3021680e813fe2bb6dccf9b22cd36fcb8c8f4b9f215c6","observation_id":"caf75ea3-bdf1-4e3a-8baa-3959a97ba84a","resolution":{"observed_at":"2026-08-06T18:21:36.125549Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-08-07T10:56:05.622094Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-06T18:21:35.679516Z","title":"Speechgpt: Empowering large language models with intrinsic cross-modal conversational abilities, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.679516Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:2648512ed686c6c7d403722865045aacce854532d58f3f403a468e04dedeb88c","observation_id":"1d118186-f278-4ef3-b83f-d1e17d70892d","resolution":{"observed_at":"2026-08-06T18:21:35.679516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:21:36.513965Z","title":"Melotts: High-quality multi-lingual multi-accent text-to-speech, 2023","venue":null,"work_id":"598ae707-3bde-48ae-882f-aee32a1ecfe5","year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.741262Z"},"links":{"citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:cabf20f719819d9518348045bd7b90ef97379b2ccb6b0c335b923e2f0586ab2f","observation_id":"cb56ea3d-eefd-43cf-83ec-eeeac2f1eb28","resolution":{"observed_at":"2026-08-06T18:21:36.599299Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.07624","last_updated":"2021-06-01T05:38:50Z","snapshot_observed_at":"2026-08-08T00:40:10.990278Z","submitted_at":"2021-05-17T06:12:06Z","title":"TAT-QA: A Question Answering Benchmark on a Hybrid of Tabular and Textual Content in Finance","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.07624","snapshot_observed_at":"2026-08-06T18:21:35.837431Z","title":"Tat-qa: A question answering benchmark on a hybrid of tabular and textual content in finance, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:35.837431Z"},"links":{"cited_paper":"/paper/2105.07624","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:fb446f94c91b22a45704759b15608a93e5a3c477929adb334fb3b4fe596d91ce","observation_id":"812782b5-2b8a-4a11-9314-71c6e9c556bd","resolution":{"observed_at":"2026-08-06T18:21:35.837431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T04:59:19.398088Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 0 inbound Pith citation observations for arXiv:2507.08603."}