{"as_of":"2026-08-17T04:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:77a53530daa3c51bd846d8cba28da547c70c43dfcc38a5d56be1930557a858cf","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T22:41:21.974244Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T01:37:30.579102Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-12T18:57:28.666194Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2311.07919"},"observation_digest":"sha256:d99d077ba6b0f5c392c5f7f6bab3016f6f06bd485ff2af93aaec4476fa4bb84b","observation_id":"2bcc33b7-1fb3-4b57-ad7c-82fe00b902f9","resolution":{"observed_at":"2026-05-12T18:57:28.748529Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-11T22:41:21.974244Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.03220","last_updated":"2024-12-04T11:14:06Z","snapshot_observed_at":"2026-08-14T10:22:33.294462Z","submitted_at":"2024-12-04T11:14:06Z","title":"Survey of different Large Language Model Architectures: Trends, Benchmarks, and Challenges","version":1},"reference_index":248,"source":"pdf_text","source_observed_at":"2026-08-11T22:41:21.974244Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2412.03220"},"observation_digest":"sha256:35a8f8644069513e5c902456df0cf538a64c8106866dbcb7bab7e28349d6770b","observation_id":"13641b6c-2ae1-4619-b66e-0e9b6d6c7119","resolution":{"observed_at":"2026-08-11T22:41:21.974244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-10T18:52:42.762925Z","title":"Llasm: Large language and speech model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.10937","last_updated":"2025-01-19T04:10:53Z","snapshot_observed_at":"2026-08-14T16:00:12.691763Z","submitted_at":"2025-01-19T04:10:53Z","title":"Leveraging Chain of Thought towards Empathetic Spoken Dialogue without Corresponding Question-Answering Data","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T18:52:42.762925Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2501.10937"},"observation_digest":"sha256:187b36d1196e8b62009f3a2064bef4ecf2388c042cc74f89b7d716496e003076","observation_id":"4989efec-6c3f-4609-bfa4-00641759fab6","resolution":{"observed_at":"2026-08-10T18:52:42.762925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2503.12605","last_updated":"2025-03-23T13:47:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-16T18:39:13Z","title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","version":2},"reference_index":200,"source":"pdf_text","source_observed_at":"2026-05-15T17:18:52.996467Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2503.12605"},"observation_digest":"sha256:dd4ba4b534e89da41e1fc9a6f84b2df4692d230c0e37525ae55b70c36415d32f","observation_id":"13bc8628-fe21-4399-95c0-379f3913b227","resolution":{"observed_at":"2026-05-15T17:18:53.367960Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2504.08528","last_updated":"2026-04-07T06:11:20Z","snapshot_observed_at":"2026-08-14T11:32:49.145887Z","submitted_at":"2025-04-11T13:40:53Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-22T20:44:57.476464Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2504.08528"},"observation_digest":"sha256:0972846f5f8dfc0ba25e6628a3c46af184f37463c2a21fb12b3094331578ef5d","observation_id":"5dbef67c-0ebb-4356-b9c2-becaa5d10622","resolution":{"observed_at":"2026-05-22T20:45:07.915989Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-07T14:30:31.680423Z","title":"Llasm: Large language and speech model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-15T11:50:55.837908Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.680423Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:920e628331c9b851878edff3aec85fd7432cfd11985f188bb16c56f57c6a54d0","observation_id":"f8e0976e-a24b-468f-853e-bd45b02588c7","resolution":{"observed_at":"2026-08-07T14:30:31.680423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-06T20:59:01.315222Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01438","last_updated":"2025-07-02T07:47:28Z","snapshot_observed_at":"2026-08-16T11:18:31.698261Z","submitted_at":"2025-07-02T07:47:28Z","title":"EdgeLoRA: An Efficient Multi-Tenant LLM Serving System on Edge Devices","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T20:59:01.315222Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2507.01438"},"observation_digest":"sha256:695e83b8182bf9afc1b8faceeacea757f3d1b5f6c8d87307e5c04931e2675ba4","observation_id":"a5b6eea1-bc57-44ca-9cab-ee12f7df98f1","resolution":{"observed_at":"2026-08-06T20:59:01.315222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-06T18:21:34.713583Z","title":"Llasm: Large language and speech model, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.08603","last_updated":"2025-07-11T13:55:45Z","snapshot_observed_at":"2026-08-14T09:41:39.436098Z","submitted_at":"2025-07-11T13:55:45Z","title":"Unlocking Speech Instruction Data Potential with Query Rewriting","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T18:21:34.713583Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2507.08603"},"observation_digest":"sha256:de76b8fb8730beea01e855eb38ff0ed4f8086ccc925fe5e416f09bca34b8f45b","observation_id":"a5992823-3346-4bbb-9f37-b589bb962408","resolution":{"observed_at":"2026-08-06T18:21:34.713583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2507.23511","last_updated":"2026-05-11T14:54:52Z","snapshot_observed_at":"2026-08-15T03:45:58.298182Z","submitted_at":"2025-07-31T12:47:43Z","title":"MECAT: A Multi-Experts Constructed Benchmark for Fine-Grained Audio Understanding Tasks","version":3},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-19T02:41:52.996457Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2507.23511"},"observation_digest":"sha256:6fc8d400221b41500075a30c185039edeffeca9bfcfa944eb8544df02c8f9ac8","observation_id":"3d004f2d-1099-4593-ae43-eff200e1d2a0","resolution":{"observed_at":"2026-05-19T02:41:59.657702Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-06T05:46:03.056127Z","title":"Llasm: Large language and speech model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.01242","last_updated":"2025-08-05T05:55:00Z","snapshot_observed_at":"2026-08-15T12:32:39.627023Z","submitted_at":"2025-08-02T07:37:37Z","title":"MeshLLM: Empowering Large Language Models to Progressively Understand and Generate 3D Mesh","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T05:46:03.056127Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2508.01242"},"observation_digest":"sha256:74e1818797335e107a383f942b236dc1cdec89a93b12ee811554145421f09131","observation_id":"a119b833-5f6d-4ced-a0db-4db488a71afd","resolution":{"observed_at":"2026-08-06T05:46:03.056127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2509.03526","last_updated":"2026-05-20T03:07:46Z","snapshot_observed_at":"2026-08-15T03:55:37.990374Z","submitted_at":"2025-08-25T07:31:48Z","title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-21T22:23:52.392075Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2509.03526"},"observation_digest":"sha256:478557167af90bbc2ef9f15d9c5b22b746e9198a35449e7329957f56cb40dc87","observation_id":"a6b5faf1-ede3-4ecf-9000-e7f17fb39398","resolution":{"observed_at":"2026-05-21T22:24:23.481271Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-12T20:09:57.922788Z","title":"LLaSM: Large Language and Speech Model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.14603","last_updated":"2026-06-29T06:44:20Z","snapshot_observed_at":"2026-07-12T20:09:54.804124Z","submitted_at":"2026-04-16T04:21:32Z","title":"A Synonymous Variational Perspective on the Rate-Distortion-Perception Tradeoff","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-12T20:09:57.922788Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2604.14603"},"observation_digest":"sha256:884de56fdd862bb6c57feeeef35fa17dc6788d06a6c574eec9aa2247d420fb13","observation_id":"bcd77563-fa62-4f61-8ae6-2bf308d2e71d","resolution":{"observed_at":"2026-07-12T20:09:57.922788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2604.14604","last_updated":"2026-04-16T04:22:11Z","snapshot_observed_at":"2026-08-11T08:37:21.535936Z","submitted_at":"2026-04-16T04:22:11Z","title":"Hijacking Large Audio-Language Models via Context-Agnostic and Imperceptible Auditory Prompt Injection","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T11:32:10.126062Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2604.14604"},"observation_digest":"sha256:7964f1d08c981142558e69a517b477ead54e178145ab0ce70dc95e8aa3236a03","observation_id":"b56a350d-6d71-441f-a73c-5883d90de0f8","resolution":{"observed_at":"2026-05-10T11:35:18.863265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2606.09366","last_updated":"2026-06-08T11:38:40Z","snapshot_observed_at":"2026-08-15T23:47:57.166790Z","submitted_at":"2026-06-08T11:38:40Z","title":"Is Text All You Need? Text as a Universal Information Bottleneck for Speech LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T16:26:32.594986Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2606.09366"},"observation_digest":"sha256:d4f9d19db801a12950ce66e4879f2a9bfeb99ee28e839ff73d227535e1ec7552","observation_id":"c55f5e85-e411-4c6e-b2e9-10faaf5f9493","resolution":{"observed_at":"2026-07-03T01:37:30.580557Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":"2308.15930","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-07-03T01:37:30.579102Z","title":"LLaSM: Large language and speech model.arXiv preprint arXiv:2308.15930","venue":null,"work_id":"2959c7f3-e690-4086-b347-09ba2e8f3989","year":2023},"citing_paper":{"arxiv_id":"2606.18273","last_updated":"2026-06-05T11:38:30Z","snapshot_observed_at":"2026-08-12T18:50:06.738816Z","submitted_at":"2026-06-05T11:38:30Z","title":"Continuous Audio Thinking for Large Audio Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T21:52:49.839901Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2606.18273"},"observation_digest":"sha256:eb1e28d86e74e87ce74700a62413597a32072c7083faa1c2ac6d250cd92865ec","observation_id":"8a0cae49-a0c4-4d5b-b531-85e456ba98c8","resolution":{"observed_at":"2026-07-02T17:47:17.971674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2308.15930/citation-record","integrity":"/paper/2308.15930/integrity","json":"/paper/2308.15930/citation-record.json","paper":"/paper/2308.15930"},"outbound":[],"paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T15:04:39.507449Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 15 inbound Pith citation observations for arXiv:2308.15930."}