{"as_of":"2026-08-10T09:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5a0cb8012a102931c558928517542fafb1ac76d1f7310ff8c6434722c25b97e3","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:30:33.802965Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:30:30.787397Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T15:18:32.647414Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.18644","snapshot_observed_at":"2026-08-07T14:30:30.787397Z","title":"Hope you<Interleaved Speech>","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:30.787397Z"},"links":{"cited_paper":"/paper/2505.18644","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:8a64b30cf77f0795af766e2c235be5f9aba8f2abc1f7a0691acc3b55d647b2e1","observation_id":"29bcc073-b6fb-49c1-8391-64c9f9a48992","resolution":{"observed_at":"2026-08-07T14:30:30.787397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"cited_work":{"arxiv_id":"2505.18644","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.18644","snapshot_observed_at":"2026-07-03T15:18:32.647414Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behav- ior Imitation and Speech-Text Interleaving","venue":null,"work_id":"85886815-17a2-4bb8-a94a-64535e9a3824","year":2025},"citing_paper":{"arxiv_id":"2509.14804","last_updated":"2025-09-18T09:59:55Z","snapshot_observed_at":"2026-08-07T14:31:44.422604Z","submitted_at":"2025-09-18T09:59:55Z","title":"Towards Building Speech Large Language Models for Multitask Understanding in Low-Resource Languages","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T16:27:37.596817Z"},"links":{"cited_paper":"/paper/2505.18644","citing_paper":"/paper/2509.14804"},"observation_digest":"sha256:7251b9e69a0c86ad92837fc0c8c2e3c67c6887fb1f94ff440317cbf8dfda420e","observation_id":"7287100f-cfe4-4531-b7de-b1e9b85c8023","resolution":{"observed_at":"2026-05-18T16:31:37.195229Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"cited_work":{"arxiv_id":"2505.18644","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.18644","snapshot_observed_at":"2026-07-03T15:18:32.647414Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behav- ior Imitation and Speech-Text Interleaving","venue":null,"work_id":"85886815-17a2-4bb8-a94a-64535e9a3824","year":2025},"citing_paper":{"arxiv_id":"2607.01733","last_updated":"2026-07-02T05:42:01Z","snapshot_observed_at":"2026-08-06T21:21:57.811880Z","submitted_at":"2026-07-02T05:42:01Z","title":"Rethinking Speech-LLM Integration for ASR: Effective Joint Speech-Text Training by Interleaving","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-03T15:17:52.966144Z"},"links":{"cited_paper":"/paper/2505.18644","citing_paper":"/paper/2607.01733"},"observation_digest":"sha256:959fe8c7bc88ae3052d8c114cf0075d2ac4d94078adb4bb0b2d8e54715a27584","observation_id":"ed1c8b03-cc89-4e2c-9120-56f69f53ecad","resolution":{"observed_at":"2026-07-03T15:18:32.648961Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.18644/citation-record","integrity":"/paper/2505.18644/integrity","json":"/paper/2505.18644/citation-record.json","paper":"/paper/2505.18644"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:35.722958Z","title":null,"venue":null,"work_id":"c11feebe-97da-4f4a-9c88-bb69bb073a22","year":null},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:30.765642Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:4c88866b56a1b5cc30e04292cf0a03e5befaa440d17f7dea3013d16ba5a7bb8d","observation_id":"d6f7a767-4caf-45ff-b453-1200e26d4611","resolution":{"observed_at":"2026-08-07T14:30:35.801501Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.18644","snapshot_observed_at":"2026-08-07T14:30:30.787397Z","title":"Hope you<Interleaved Speech>","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:30.787397Z"},"links":{"cited_paper":"/paper/2505.18644","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:8a64b30cf77f0795af766e2c235be5f9aba8f2abc1f7a0691acc3b55d647b2e1","observation_id":"29bcc073-b6fb-49c1-8391-64c9f9a48992","resolution":{"observed_at":"2026-08-07T14:30:30.787397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:35.459201Z","title":"To address this gap, we draw inspiration from estab- lished evaluation methods in NLP [15, 16]","venue":null,"work_id":"cf6555dd-f13e-41bb-bdbb-42f4adfe5b27","year":null},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:30.917143Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:4851a7088ffdcf79b8737872cec25bc3124a3cef04319ccb2ec12b8c9653b68b","observation_id":"efcfb8a1-59cb-4dd6-85fe-dd4e942a6caf","resolution":{"observed_at":"2026-08-07T14:30:35.600996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:35.296256Z","title":"This task evaluates both the instruction following and reasoning abilities of SLLMs","venue":null,"work_id":"d1f63dbc-f3bd-479f-8004-a927d0675f4b","year":null},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.039709Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b5b215bb70f435f13abafcbd899ddf15cd69facded4976de4a3699f193d340aa","observation_id":"776bc234-ea5e-482a-96f5-3c403db46458","resolution":{"observed_at":"2026-08-07T14:30:35.372184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:35.097617Z","title":null,"venue":null,"work_id":"0ef8e2a4-6af6-411e-aa35-39c11f4cc5b0","year":null},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.169989Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:6a7cf03dc283606dd7a8633bcdfff6be4fa2c38e96afc55e70ee6bd84766e30c","observation_id":"9c694860-f940-4deb-abbe-bda9cda12159","resolution":{"observed_at":"2026-08-07T14:30:35.179258Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:34.891218Z","title":null,"venue":null,"work_id":"2955f203-fd37-4f19-a00e-fd02cbcef2b7","year":null},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.303114Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:10f9f6a1022b45840a589fcc5d0fe134fc9e2ddb4156ed96a337a39dc56bafc1","observation_id":"59c2f1a1-8650-421c-b3bd-3ebf8996c151","resolution":{"observed_at":"2026-08-07T14:30:34.979967Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-07T14:30:31.425290Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.425290Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:97c3ffbf59383b387300c048623042fdefc95cb863e2ba45dcdc4284b41c23ae","observation_id":"8b715cbe-a44f-461b-90ea-9e4169c7cd5a","resolution":{"observed_at":"2026-08-07T14:30:31.425290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-07T14:30:31.517878Z","title":"Llama 2: Open foundation and fine-tuned chat models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.517878Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:7cdf7e812ff14ead3760a79bb6b082e6a5ea84a58716b54e62540108a77133ec","observation_id":"59cf28db-b5e6-4a48-9849-38e855f68801","resolution":{"observed_at":"2026-08-07T14:30:31.517878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-08-07T10:56:05.622094Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-07T14:30:31.589269Z","title":"Speechgpt: Empowering large language models with intrinsic cross-modal conversational abilities,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.589269Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:fe16e06f5f4a04ef46e1e68cbbc7753b25128131294299837dabbb71581f4b4c","observation_id":"88cc677d-846b-40eb-980f-c20e16e3560f","resolution":{"observed_at":"2026-08-07T14:30:31.589269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.15930","last_updated":"2023-09-16T06:14:54Z","snapshot_observed_at":"2026-08-05T06:18:44.577926Z","submitted_at":"2023-08-30T10:12:39Z","title":"LLaSM: Large Language and Speech Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.15930","snapshot_observed_at":"2026-08-07T14:30:31.680423Z","title":"Llasm: Large language and speech model,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.680423Z"},"links":{"cited_paper":"/paper/2308.15930","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:239b7ecc5141349ab2d3788b4c3e6c275954214a939d3073a21dd3eb7fbc57c5","observation_id":"f8e0976e-a24b-468f-853e-bd45b02588c7","resolution":{"observed_at":"2026-08-07T14:30:31.680423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07919","last_updated":"2023-12-21T10:20:42Z","snapshot_observed_at":"2026-08-07T10:17:55.688598Z","submitted_at":"2023-11-14T05:34:50Z","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07919","snapshot_observed_at":"2026-08-07T14:30:31.743024Z","title":"Qwen-audio: Advancing universal audio understand- ing via unified large-scale audio-language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.743024Z"},"links":{"cited_paper":"/paper/2311.07919","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:5a896a0bc97f869462b674240d9a5872d8ae5fbabe23633533a36a88c34d45c9","observation_id":"1ba8552a-e7fd-4b57-bbf7-b00540d94683","resolution":{"observed_at":"2026-08-07T14:30:31.743024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-07T14:30:31.863555Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.863555Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:162a5c3a4649bce2a511f30fb6e58bea7939aa56b23dfe63249d7fa620184b25","observation_id":"e3d616bc-cdc1-46f4-b0dc-41353a7b27cb","resolution":{"observed_at":"2026-08-07T14:30:31.863555Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16725","last_updated":"2024-11-05T02:24:18Z","snapshot_observed_at":"2026-07-06T19:07:46.545514Z","submitted_at":"2024-08-29T17:18:53Z","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16725","snapshot_observed_at":"2026-08-07T14:30:31.918185Z","title":"Mini-omni: Language models can hear, talk while thinking in streaming,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.918185Z"},"links":{"cited_paper":"/paper/2408.16725","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:ccab8dea4f09c91c64c349987ff3981bdb3c9b3d273d6fcb958b18cbc7e77aeb","observation_id":"b151d448-45cf-463a-84cd-0e688b996c67","resolution":{"observed_at":"2026-08-07T14:30:31.918185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:31.987984Z","title":"Language models are unsupervised multitask learners,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:31.987984Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b28cecd454be872fb725db02d0be7ba2cb8f668a1b1642199c8c1481dcf48a15","observation_id":"0c85785d-85b2-440a-9d87-e3e9af258536","resolution":{"observed_at":"2026-08-07T14:30:31.987984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:34.739643Z","title":"Multi-task learning in natu- ral language processing: An overview,","venue":null,"work_id":"da591ca8-aced-4b80-b33f-65a1d20d55aa","year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.075506Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:53647b57ed43cb2b16127950c226e5ab1b0725d3f09db879f9621beedb258c49","observation_id":"cf6e60a3-1e37-4e9e-8eb1-1857748c2ac4","resolution":{"observed_at":"2026-08-07T14:30:34.782094Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00916","last_updated":"2024-05-28T14:26:28Z","snapshot_observed_at":"2026-07-06T16:13:34.788427Z","submitted_at":"2023-09-02T11:46:05Z","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00916","snapshot_observed_at":"2026-08-07T14:30:32.148354Z","title":"Blsp: Bootstrapping language-speech pre-training via behavior alignment of continuation writing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.148354Z"},"links":{"cited_paper":"/paper/2309.00916","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:20fce88aab0c1d79d8a75acfc900bc65e7118bbb139a8489a1e61d67d82b5824","observation_id":"94f98aa1-3d84-46a6-a31c-e2f6c2a8f51a","resolution":{"observed_at":"2026-08-07T14:30:32.148354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08295","last_updated":"2025-03-24T21:06:53Z","snapshot_observed_at":"2026-07-06T18:13:55.543513Z","submitted_at":"2024-05-14T03:33:31Z","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08295","snapshot_observed_at":"2026-08-07T14:30:32.222856Z","title":"Speechverse: A large-scale generalizable audio language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.222856Z"},"links":{"cited_paper":"/paper/2405.08295","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b488d351d9cb13da641317e622bc923cb348400074d8c5714dcbda3886e4b21b","observation_id":"1e9316b4-b1fd-426f-8937-3caf555a31f6","resolution":{"observed_at":"2026-08-07T14:30:32.222856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.17607","last_updated":"2024-12-02T16:13:24Z","snapshot_observed_at":"2026-08-09T12:37:18.065852Z","submitted_at":"2024-11-26T17:19:09Z","title":"Scaling Speech-Text Pre-training with Synthetic Interleaved Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.17607","snapshot_observed_at":"2026-08-07T14:30:32.280525Z","title":"Scaling speech-text pre-training with synthetic interleaved data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.280525Z"},"links":{"cited_paper":"/paper/2411.17607","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:6f5106fdf3c3a25b61a35a6fc33d8b49e7baacffaaff21a5f3f33c7a0a559d8a","observation_id":"85c9fb60-1c6d-4d7c-b3dc-f1d49cfd4c94","resolution":{"observed_at":"2026-08-07T14:30:32.280525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05755","last_updated":"2024-10-18T19:18:41Z","snapshot_observed_at":"2026-08-08T16:34:49.867190Z","submitted_at":"2024-02-08T15:39:32Z","title":"Spirit LM: Interleaved Spoken and Written Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05755","snapshot_observed_at":"2026-08-07T14:30:32.356065Z","title":"Spirit-lm: Interleaved spoken and written language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.356065Z"},"links":{"cited_paper":"/paper/2402.05755","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:399fa85f664d15adce16041b8de86644db777d891739dc49a26c84e7b73f784f","observation_id":"d8b4e2b3-bb32-4a29-aea2-75bef8783034","resolution":{"observed_at":"2026-08-07T14:30:32.356065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12488","last_updated":"2024-12-24T08:58:16Z","snapshot_observed_at":"2026-07-06T16:09:48.222394Z","submitted_at":"2023-08-24T01:17:16Z","title":"GPTEval: A Survey on Assessments of ChatGPT and GPT-4","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12488","snapshot_observed_at":"2026-08-07T14:30:32.434957Z","title":"Gpteval: A survey on assessments of chatgpt and gpt-4,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.434957Z"},"links":{"cited_paper":"/paper/2308.12488","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:91660ddf029337175c9cc6716802e0783e827dd3dc0bada10ccadcd1ad6ca65a","observation_id":"3e1c4023-e4d4-4822-82c5-7a08c715b20a","resolution":{"observed_at":"2026-08-07T14:30:32.434957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-07T14:30:32.525199Z","title":"Instruction-following evaluation for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.525199Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b0be1be15484501be6cec04caff34aaabaaf7947e05d88e0af54040df2e44595","observation_id":"aa2aa27e-4c85-4a5f-89d7-b37a0d651418","resolution":{"observed_at":"2026-08-07T14:30:32.525199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01383","last_updated":"2025-05-14T06:05:53Z","snapshot_observed_at":"2026-08-08T11:17:56.985482Z","submitted_at":"2024-02-02T13:06:35Z","title":"LLM-based NLG Evaluation: Current Status and Challenges","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01383","snapshot_observed_at":"2026-08-07T14:30:32.605080Z","title":"Llm-based nlg evaluation: Current status and challenges,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.605080Z"},"links":{"cited_paper":"/paper/2402.01383","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:27bb1a0987bd7aa0020307aaee98cec2ff3acc44c93dba6b170c181e2b05e3b0","observation_id":"d249389f-0cbc-4361-a489-06d8187191c4","resolution":{"observed_at":"2026-08-07T14:30:32.605080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10937","last_updated":"2025-01-19T04:10:53Z","snapshot_observed_at":"2026-07-06T20:22:59.218985Z","submitted_at":"2025-01-19T04:10:53Z","title":"Leveraging Chain of Thought towards Empathetic Spoken Dialogue without Corresponding Question-Answering Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.10937","snapshot_observed_at":"2026-08-07T14:30:32.689968Z","title":"Leveraging chain of thought towards empathetic spoken dialogue without corresponding question-answering data,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.689968Z"},"links":{"cited_paper":"/paper/2501.10937","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:c4121b5bca0836f8528eebfb061488248b92e73a5812c0a2dc1e4c645d1ff649","observation_id":"bb23efa7-939b-4348-b0bf-242f3a858245","resolution":{"observed_at":"2026-08-07T14:30:32.689968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:32.779482Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.779482Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:abcf98f9548f43aa9d4ebc6d5ccee2fd7d4c1540f251ca0d90676910466fd01e","observation_id":"03aa659d-143e-4f58-a6c3-dd19882ba531","resolution":{"observed_at":"2026-08-07T14:30:32.779482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16199","last_updated":"2024-09-18T23:54:36Z","snapshot_observed_at":"2026-08-06T06:36:02.994951Z","submitted_at":"2023-03-28T17:59:12Z","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.16199","snapshot_observed_at":"2026-08-07T14:30:32.843707Z","title":"Llama-adapter: Efficient fine-tuning of language models with zero-init attention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.843707Z"},"links":{"cited_paper":"/paper/2303.16199","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b3ce3741ad476c1ac274b4bcb78b384cba37eb4ba877fd8692da4513cfabc7cd","observation_id":"91e79bc3-cbb2-4c8f-91c3-2f9b851bbdec","resolution":{"observed_at":"2026-08-07T14:30:32.843707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:32.968567Z","title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:32.968567Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:682b4ee5d4389d9a66e4c15135fcdbdbf65bf7c7267cb74ee6d43a26d14cdd8c","observation_id":"a6b0e2ac-f3cc-458b-b5f1-7a05d1bccbac","resolution":{"observed_at":"2026-08-07T14:30:32.968567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.10117","last_updated":"2024-12-25T11:54:03Z","snapshot_observed_at":"2026-08-09T04:36:59.879757Z","submitted_at":"2024-12-13T12:59:39Z","title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.10117","snapshot_observed_at":"2026-08-07T14:30:33.069465Z","title":"Cosyvoice 2: Scalable stream- ing speech synthesis with large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.069465Z"},"links":{"cited_paper":"/paper/2412.10117","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:7a4141e511963fbd120cad79d5bac026c04ca91e672b77d67cbc08882692bdfa","observation_id":"1287b22a-1773-48bc-9f46-c80ba15d3a0d","resolution":{"observed_at":"2026-08-07T14:30:33.069465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:34.540366Z","title":"Hello gpt-4o,","venue":null,"work_id":"86111f51-161f-43ba-8a2e-8d3ae3bea50b","year":2024},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.147124Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:1af1e955fb14f5dda23b6f75fa95f7c2ba2e651decfed6b2160eb4b010bb3405","observation_id":"41aae15f-3d3b-484f-9192-aa32e66f3a62","resolution":{"observed_at":"2026-08-07T14:30:34.600778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T14:30:33.219850Z","title":"Train- ing verifiers to solve math word problems,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.219850Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:b75582f725d693c1bcfb3374d59de390077be24ceb6015c9c5e829c85b54f396","observation_id":"4768dfc9-f8be-469d-aa4a-351cb8402b54","resolution":{"observed_at":"2026-08-07T14:30:33.219850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T14:30:33.298274Z","title":"Measuring mathemat- ical problem solving with the math dataset,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.298274Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:f6d3fb82563a526fe9b99fd1b5c17ecc149f56dc9bdab7ce58a7bf295f5c56a8","observation_id":"dc274852-8a96-486e-a07b-227f53a54fc5","resolution":{"observed_at":"2026-08-07T14:30:33.298274Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:33.363830Z","title":"Lib- rispeech: an asr corpus based on public domain audio books,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.363830Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:0e5b962d238ffd93009f0efa28ff9416b607699f6e15e6d5a590a7ea1c528b95","observation_id":"c638f1c8-ab81-459f-9d03-0ceda5bec962","resolution":{"observed_at":"2026-08-07T14:30:33.363830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1805.10190","last_updated":"2018-12-06T16:34:25Z","snapshot_observed_at":"2026-08-05T10:58:51.399386Z","submitted_at":"2018-05-25T15:04:17Z","title":"Snips Voice Platform: an embedded Spoken Language Understanding system for private-by-design voice interfaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.10190","snapshot_observed_at":"2026-08-07T14:30:33.466627Z","title":"Snips voice platform: an embedded spoken language under- standing system for private-by-design voice interfaces,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.466627Z"},"links":{"cited_paper":"/paper/1805.10190","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:927359bc5bcff95063152360a69ca9f7c20c412a5c864e8cd48357456bb898ea","observation_id":"77994d98-1949-4ee6-87e6-d9a144cdba15","resolution":{"observed_at":"2026-08-07T14:30:33.466627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-07T14:30:33.562288Z","title":"Speech model pre-training for end-to-end spoken language understanding,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.562288Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:37ffe0442159f759df8b99563704b0127c28dea3c7d7fcbca31f49c7ce1c54c7","observation_id":"5cc71a69-6342-4500-8b2d-4106f84f54fa","resolution":{"observed_at":"2026-08-07T14:30:33.562288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.06670","last_updated":"2020-03-05T20:37:08Z","snapshot_observed_at":"2026-08-05T11:17:11.092255Z","submitted_at":"2019-12-13T19:22:44Z","title":"Common Voice: A Massively-Multilingual Speech Corpus","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.06670","snapshot_observed_at":"2026-08-07T14:30:33.640309Z","title":"Common voice: A massively-multilingual speech corpus,","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.640309Z"},"links":{"cited_paper":"/paper/1912.06670","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:11af25159d556e8dba171b8290ba0d4c4dfff1ce39cfd5546c5942439906994d","observation_id":"e30882b3-e03d-401d-80b8-addefa9bd630","resolution":{"observed_at":"2026-08-07T14:30:33.640309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:30:34.358196Z","title":"Covost 2 and massively multilingual speech translation","venue":null,"work_id":"c1ab4426-588a-4c0d-be5b-546e9cbc11a5","year":2021},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.717606Z"},"links":{"citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:ab098acae32da59e88689a754d8cae1750278f92b3426fbbf2789effbe7d6e70","observation_id":"1cd6abe1-9626-40db-ad84-8e076c09b29c","resolution":{"observed_at":"2026-08-07T14:30:34.436593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-07T14:30:33.802965Z","title":"Robust speech recognition via large- scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.802965Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:8d0ef3c5ad73cece881d060961a2335952741d5540eb19ae9ee88294e806756d","observation_id":"9ab42704-7e69-4852-a0a3-61943c0b3599","resolution":{"observed_at":"2026-08-07T14:30:33.802965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":30,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 3 inbound Pith citation observations for arXiv:2505.18644."}