{"as_of":"2026-08-20T04:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:71160255587c6d2316bbe178dbc3d9e6c5fdcf675e3a5dc00558ad0f68f96dc0","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T04:44:25.985859Z","state":"measured"},{"denominator":50,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":50,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T21:52:53.386966Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T20:45:08.147897Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"cited_work":{"arxiv_id":"2412.01145","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.01145","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"AlignFormer: Modality matching can achieve better zero-shot instruction-following speech-LLM.arXiv preprint arXiv:2412.01145","venue":null,"work_id":"f6519f2c-6535-4617-9f5a-1611717d8de0","year":null},"citing_paper":{"arxiv_id":"2504.08528","last_updated":"2026-04-07T06:11:20Z","snapshot_observed_at":"2026-08-14T11:32:49.145887Z","submitted_at":"2025-04-11T13:40:53Z","title":"On The Landscape of Spoken Language Models: A Comprehensive Survey","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T20:44:57.476464Z"},"links":{"cited_paper":"/paper/2412.01145","citing_paper":"/paper/2504.08528"},"observation_digest":"sha256:357ca1a8ff3c6fbcb8a0350006a4c84f6e8292a258aae7cdd038c0ec49a49b17","observation_id":"365b54bc-1661-477d-8383-51bc92147390","resolution":{"observed_at":"2026-05-22T20:45:08.151044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.01145","snapshot_observed_at":"2026-08-15T21:52:53.386966Z","title":"Alignformer: Modality matching can achieve better zero-shot instruction-following speech-llm,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.08699","last_updated":"2025-05-14T02:10:29Z","snapshot_observed_at":"2026-08-15T23:01:43.367826Z","submitted_at":"2025-05-13T15:58:57Z","title":"Granite-speech: open-source speech-aware LLMs with strong English ASR capabilities","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T21:52:53.386966Z"},"links":{"cited_paper":"/paper/2412.01145","citing_paper":"/paper/2505.08699"},"observation_digest":"sha256:615f67aa5a5bbb5fcb5084394035154b69c5d825b93960aef9fdd8903b1ad84a","observation_id":"0b04b37b-d2f0-4758-a944-eb730058e271","resolution":{"observed_at":"2026-08-15T21:52:53.386966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.01145/citation-record","integrity":"/paper/2412.01145/integrity","json":"/paper/2412.01145/citation-record.json","paper":"/paper/2412.01145"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.692958Z","title":"Language models are few-shot learners,","venue":null,"work_id":"da212e27-b83e-4dd3-8f8b-140ad6bfccdd","year":2020},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.784371Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:847960c61087142143b12cde96c50a5d4353ef39e05e077a4ce8fa788d040efd","observation_id":"985a90aa-debf-49ca-be33-294fffdccfa5","resolution":{"observed_at":"2026-08-12T04:44:26.697786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T04:44:25.790256Z","title":"Gpt-4 technical report,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.790256Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:bb83ccadc03612ba699b3cea6ceb80a955892ef2f2e9fdc7b64de250d0c193a0","observation_id":"716d7f8d-3a7c-4636-b8df-37eb5a2e534d","resolution":{"observed_at":"2026-08-12T04:44:25.790256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-12T04:44:25.795189Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.795189Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:d1a01f5ba73f58b4aa431d0fd94c8285356d2650767c1814f7119fc98c3fda0b","observation_id":"0cb7d0de-fd21-46ad-abff-27c6cbd21103","resolution":{"observed_at":"2026-08-12T04:44:25.795189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.670285Z","title":"Self-instruct: Aligning language models with self- generated instructions,","venue":null,"work_id":"e2878628-19fa-4bb7-9299-0967d23cb0e8","year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.801322Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:8b45dbcec7a61d2bf6017c317384b8d9cf571d9c2ea1f2797bfab68ab79d081f","observation_id":"c8df4194-3025-4b92-abd6-a0202eeb08ec","resolution":{"observed_at":"2026-08-12T04:44:26.680275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:25.806894Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.806894Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:2dc850a7730e6f7a67da44369e091bc0e950cb8e91bb0b6f25f9eac73b8b106e","observation_id":"3cdce924-0448-4e9e-bce3-0b7308d2b00e","resolution":{"observed_at":"2026-08-12T04:44:25.806894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:25.811299Z","title":"Direct preference optimization: Your language model is secretly a reward model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.811299Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:0df39d61e1cfae73cf3a41d5aa1330b1770491cf4e780893fc1ea2cb2dd17fdc","observation_id":"329c671c-df29-4ec5-92ad-516a771e5422","resolution":{"observed_at":"2026-08-12T04:44:25.811299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-12T04:44:25.816937Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.816937Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:3f3241b770ee140b5d8a187e314eca8361706fbd71557b6d2fb3c053f3fe7303","observation_id":"3a2cc399-9573-4da9-aefa-2645b242341f","resolution":{"observed_at":"2026-08-12T04:44:25.816937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00774","last_updated":"2024-12-08T05:41:56Z","snapshot_observed_at":"2026-08-16T13:03:42.737566Z","submitted_at":"2024-11-01T17:59:51Z","title":"Freeze-Omni: A Smart and Low Latency Speech-to-speech Dialogue Model with Frozen LLM","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00774","snapshot_observed_at":"2026-08-12T04:44:25.821357Z","title":"Freeze-omni: A smart and low latency speech-to-speech dialogue model with frozen llm,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.821357Z"},"links":{"cited_paper":"/paper/2411.00774","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:fe0b6e3f1bb3bebec92eb0dcab4cdb5c9ed9dea8ec6619ea996917604071464c","observation_id":"06b2145c-2d85-4191-bde8-1c5a2f153906","resolution":{"observed_at":"2026-08-12T04:44:25.821357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11190","last_updated":"2024-11-05T02:27:57Z","snapshot_observed_at":"2026-08-16T13:09:22.197235Z","submitted_at":"2024-10-15T02:10:45Z","title":"Mini-Omni2: Towards Open-source GPT-4o with Vision, Speech and Duplex Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11190","snapshot_observed_at":"2026-08-12T04:44:25.825642Z","title":"Mini-omni2: Towards open-source gpt-4o with vi- sion, speech and duplex capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.825642Z"},"links":{"cited_paper":"/paper/2410.11190","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:481a0dfa37377477d71fad27b77268c785fbbd80f11f90ad99b78b0eec4d5845","observation_id":"36998284-953c-477a-95da-df47762a1d60","resolution":{"observed_at":"2026-08-12T04:44:25.825642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18138","last_updated":"2024-11-27T08:38:57Z","snapshot_observed_at":"2026-08-18T02:53:18.978175Z","submitted_at":"2024-11-27T08:38:57Z","title":"SALMONN-omni: A Codec-free LLM for Full-duplex Speech Understanding and Generation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18138","snapshot_observed_at":"2026-08-12T04:44:25.829811Z","title":"Salmonn-omni: A codec-free llm for full-duplex speech understanding and generation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.829811Z"},"links":{"cited_paper":"/paper/2411.18138","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:034c3892fd19acd5bd6ef82236959cf870c2a1029ef2c84e50de3fa1643c0e54","observation_id":"8c733e4d-e0fc-41df-9404-1b8e19304a9a","resolution":{"observed_at":"2026-08-12T04:44:25.829811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-08-19T05:23:55.031463Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-12T04:44:25.834345Z","title":"Speechgpt: Empowering large language models with intrinsic cross- modal conversational abilities,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.834345Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:58d5664878fcbb72743b71e152098988e65aa583438352ae27466f28062a441e","observation_id":"19c436fa-ff2f-4d68-8e51-6a9e950764e1","resolution":{"observed_at":"2026-08-12T04:44:25.834345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04673","last_updated":"2024-07-03T02:38:03Z","snapshot_observed_at":"2026-08-16T14:54:18.806254Z","submitted_at":"2023-10-07T03:17:59Z","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04673","snapshot_observed_at":"2026-08-12T04:44:25.838344Z","title":"Lauragpt: Listen, attend, understand, and regenerate audio with gpt,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.838344Z"},"links":{"cited_paper":"/paper/2310.04673","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:aacd689545559729de8d5276668e8c69f5b78e9a9a71bef7583effc4d0f7cd89","observation_id":"fedfc8f9-0c10-4bc1-80ab-eae0ffdfc593","resolution":{"observed_at":"2026-08-12T04:44:25.838344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08846","last_updated":"2024-02-13T23:25:04Z","snapshot_observed_at":"2026-08-16T14:18:48.581750Z","submitted_at":"2024-02-13T23:25:04Z","title":"An Embarrassingly Simple Approach for LLM with Strong ASR Capacity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08846","snapshot_observed_at":"2026-08-12T04:44:25.841973Z","title":"An embarrassingly simple approach for llm with strong asr capacity,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.841973Z"},"links":{"cited_paper":"/paper/2402.08846","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:c81742583c69a786e56c0be83f7d2fe2fbdf3dd8d29229eece3150f1e19f6cea","observation_id":"e60d1685-300a-4a71-b8d3-9d17e046fbbc","resolution":{"observed_at":"2026-08-12T04:44:25.841973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.637361Z","title":"SALMONN: towards generic hearing abilities for large language models,","venue":null,"work_id":"8528feee-c396-4464-a685-a81cb88aaaf3","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.845827Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:f4c4c986ef4dca4e633812fca0d89fa2a5e9dcbb8e9db265960922fe4f2f6c99","observation_id":"c4f4b24e-79ee-4853-945e-ef4672dfdf7d","resolution":{"observed_at":"2026-08-12T04:44:26.641519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10759","last_updated":"2024-07-15T14:38:09Z","snapshot_observed_at":"2026-08-14T01:27:16.843576Z","submitted_at":"2024-07-15T14:38:09Z","title":"Qwen2-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10759","snapshot_observed_at":"2026-08-12T04:44:25.849796Z","title":"Qwen2-audio technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.849796Z"},"links":{"cited_paper":"/paper/2407.10759","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:f1d4e7303aabd69fe6f2e040ae1e3b40b74246660f8eb340a9b91f07c172e863","observation_id":"5997242c-74a7-4943-81cf-1c028e5d6d06","resolution":{"observed_at":"2026-08-12T04:44:25.849796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.621578Z","title":"On decoder-only architecture for speech- to-text and large language model integration,","venue":null,"work_id":"0169cf8b-cc8c-4372-b667-eaa4a7697c73","year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.854425Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:88f830de80fd0848a0c071abed91e622b028d7f3f5d41541ad1080978952070e","observation_id":"181cedd0-34f3-49bf-a163-cc409242e82e","resolution":{"observed_at":"2026-08-12T04:44:26.626933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.602724Z","title":"COSMIC: data efficient instruction-tuning for speech in-context learn- ing,","venue":null,"work_id":"67713b6c-f547-4899-9de0-889c574712ff","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.858028Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:37a2fc67217ec2542a3d24459497d4abd4145796553885a6ef5c70bb9181367b","observation_id":"782fcbb5-d293-44d4-b352-6e7db367d78f","resolution":{"observed_at":"2026-08-12T04:44:26.610104Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.584239Z","title":"Prompting large language models with speech recognition abilities,","venue":null,"work_id":"b2b9e4b9-d186-43b7-87e5-e67e22184f2f","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.862163Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:5a0939b94e3f0863213289fd28c18a6f698820046b0a7cd035f5f0e85e30260b","observation_id":"e622f2e9-4fe5-42dc-bd14-35fab6d28c79","resolution":{"observed_at":"2026-08-12T04:44:26.590364Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.08295","last_updated":"2025-03-24T21:06:53Z","snapshot_observed_at":"2026-08-16T13:52:57.510888Z","submitted_at":"2024-05-14T03:33:31Z","title":"SpeechVerse: A Large-scale Generalizable Audio Language Model","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.08295","snapshot_observed_at":"2026-08-12T04:44:25.866877Z","title":"Speechverse: A large-scale generalizable audio language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.866877Z"},"links":{"cited_paper":"/paper/2405.08295","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:71a96a9cfa6794a0fb9fe2f0501ce5269606a5a161aa2757320b619cd8139fc6","observation_id":"7df69f19-523c-44be-8936-bd2a9a07ecf9","resolution":{"observed_at":"2026-08-12T04:44:25.866877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:25.870796Z","title":"High fidelity neural audio compression,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.870796Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:de96d230867b2b3072c4b6d25594c3563d2657cdaf78b9d8f3ad231f3718fd91","observation_id":"97d59d3c-1b84-4e41-85d6-add42654e0e4","resolution":{"observed_at":"2026-08-12T04:44:25.870796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.559887Z","title":"High- fidelity audio compression with improved rvqgan,","venue":null,"work_id":"6fefcfe7-4444-4fc5-b478-7bca0ab7f621","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.874350Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:b0bf187d540c811a8f44ce67c77f1f5ca1d745a4ed8c8a092bbb809477f8bf96","observation_id":"623da227-e956-4251-886e-484a7e084a54","resolution":{"observed_at":"2026-08-12T04:44:26.563623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.04675","last_updated":"2024-07-10T09:01:17Z","snapshot_observed_at":"2026-08-16T13:36:42.713695Z","submitted_at":"2024-07-05T17:38:03Z","title":"Seed-ASR: Understanding Diverse Speech and Contexts with LLM-based Speech Recognition","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.04675","snapshot_observed_at":"2026-08-12T04:44:25.878311Z","title":"Seed-asr: Understanding diverse speech and con- texts with llm-based speech recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.878311Z"},"links":{"cited_paper":"/paper/2407.04675","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:d1d8555a8c6e71eaa04df4b7b0586852df41323878ea25313482038842e3e10e","observation_id":"818a281e-6dc3-40a4-a1bd-5befccdef7f2","resolution":{"observed_at":"2026-08-12T04:44:25.878311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.548797Z","title":"Wavllm: Towards robust and adaptive speech large language model,","venue":null,"work_id":"0a9ab1d5-697f-4e2e-a328-2b822b16a45b","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.882930Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:69807817fefc0d0c20315b9fcb15a21bb24ec1d1c22334df521f66482aaf4cce","observation_id":"873c4ec5-451a-4960-a1e5-be8b46640b66","resolution":{"observed_at":"2026-08-12T04:44:26.553469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.515065Z","title":"Audiochatllama: Towards general-purpose speech abilities for llms,","venue":null,"work_id":"a71006e0-166c-4707-97ea-2180cd8d089c","year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.889898Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:11a84aa2d377dbe12061902bd965ff5782db8c40702079526cf1d305b620ce6d","observation_id":"da070468-5174-4080-9257-9d6e12c61ff6","resolution":{"observed_at":"2026-08-12T04:44:26.519938Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00916","last_updated":"2024-05-28T14:26:28Z","snapshot_observed_at":"2026-08-19T03:28:57.113432Z","submitted_at":"2023-09-02T11:46:05Z","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.00916","snapshot_observed_at":"2026-08-12T04:44:25.893673Z","title":"Blsp: Bootstrapping language-speech pre-training via behavior align- ment of continuation writing,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.893673Z"},"links":{"cited_paper":"/paper/2309.00916","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:bcb02695c82d8bf6ed6f3d3c646b1c1c96d492a536baecf2cb581963bb4c27c2","observation_id":"b6e663c6-107c-432c-9304-37c183a30bc2","resolution":{"observed_at":"2026-08-12T04:44:25.893673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.20007","last_updated":"2025-01-27T08:46:09Z","snapshot_observed_at":"2026-08-18T07:57:26.404648Z","submitted_at":"2024-09-30T07:01:21Z","title":"DeSTA2: Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.20007","snapshot_observed_at":"2026-08-12T04:44:25.897763Z","title":"Developing instruction-following speech language model without speech instruction-tuning data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.897763Z"},"links":{"cited_paper":"/paper/2409.20007","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:da1279a4aa04bad6240f82b9d734875e72a42aebe2c97439e854cb35e50f06d6","observation_id":"44f31491-4a10-408e-9d3f-60af5e18d21c","resolution":{"observed_at":"2026-08-12T04:44:25.897763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01162","last_updated":"2025-05-19T22:01:39Z","snapshot_observed_at":"2026-08-16T13:13:32.629020Z","submitted_at":"2024-10-02T01:32:47Z","title":"Frozen Large Language Models Can Perceive Paralinguistic Aspects of Speech","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01162","snapshot_observed_at":"2026-08-12T04:44:25.901872Z","title":"Frozen large language models can perceive paralinguistic aspects of speech,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.901872Z"},"links":{"cited_paper":"/paper/2410.01162","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:0eceeb4b1119e1f6d1f1847ceb1ebe5ea89f3360504911f7bcb153b6f0ac0ccc","observation_id":"36749f60-17f5-410b-810f-9da282938ede","resolution":{"observed_at":"2026-08-12T04:44:25.901872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.502292Z","title":"Wav2Prompt: End-to-end speech prompt learning and task-based fine-tuning for text-based LLMs,","venue":null,"work_id":"5f344cab-597c-4fe0-b8d5-f5276a160c9d","year":2025},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.905887Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:4214e9adbd0680ef9d0eaf9b1a3b8f055f58cab97796f9027c41c273f87f367f","observation_id":"6c5b8b81-e1a0-439a-b105-2433776bea43","resolution":{"observed_at":"2026-08-12T04:44:26.506587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-08-17T03:25:04.404839Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-12T04:44:25.909775Z","title":"Phi-3 technical report: A highly capable language model locally on your phone,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.909775Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:9446dee0a43a4be5379772b47a359999b76e9cd85885578275df69919dc7dc4f","observation_id":"92414057-f770-43da-ad24-19aa9404c3a0","resolution":{"observed_at":"2026-08-12T04:44:25.909775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-15T22:57:45.773661Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-12T04:44:25.914477Z","title":"Phi-4-mini technical report: Compact yet powerful multimodal language models via mixture- of-loras,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.914477Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:b9bc3c7dec410c994b129c1461fe4e9c03f70f1c4db82da03d003d3733d8c019","observation_id":"36333554-2450-44f8-8090-6b818c617999","resolution":{"observed_at":"2026-08-12T04:44:25.914477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.485924Z","title":"Connec- tionist temporal classification: labelling unsegmented sequence data with recurrent neural networks,","venue":null,"work_id":"3d5d6a88-5fe1-4cf7-80a7-f04ba904d196","year":2006},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.919090Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:8081397ee1146da1c761961787c00679010f6af9cdfd4ca7c44291bc43627056","observation_id":"65626896-ee5e-4025-a331-63ae60dc2ac7","resolution":{"observed_at":"2026-08-12T04:44:26.490200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.468975Z","title":"CASS-NAT: CTC alignment- based single step non-autoregressive transformer for speech recognition,","venue":null,"work_id":"3642a174-6bf4-460c-9571-82599fa75754","year":2021},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.923067Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:ee1c621ad8d2235fd86fd0b80bc7d43cf87786123e51bbd3d1eed942079c10e8","observation_id":"71511e59-350c-4d5c-a4fd-e8c8a0a96a34","resolution":{"observed_at":"2026-08-12T04:44:26.473692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.457024Z","title":"Unienc-cassnat: An encoder-only non-autoregressive asr for speech ssl models,","venue":null,"work_id":"98ebb9bb-dde2-46bf-b6a5-32dfe32c134a","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.927088Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:e221c10f909d430f25c313c6b8690f49936b10930d6ecb4c1346326a9b8e9b19","observation_id":"e5e6650f-2ee5-4b1c-ab49-d42a5d3749a0","resolution":{"observed_at":"2026-08-12T04:44:26.461458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.445204Z","title":"Ctc-based compression for direct speech translation,","venue":null,"work_id":"408f24ca-328e-45f6-9e9d-f673ebedecb0","year":2021},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.932204Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:8e52d7dda521a7ee92db41bed80148fc79b12dce447b8838916a615c90b6a3fb","observation_id":"9c2802fb-909d-4e2d-b02f-3ac59ab22d71","resolution":{"observed_at":"2026-08-12T04:44:26.449574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.431165Z","title":"CTC-GMM: CTC guided modality matching for fast and accurate streaming speech translation,","venue":null,"work_id":"34d61fb6-e11a-4603-9871-629c75beab9e","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.935963Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:7179f7fc3973ac7331561c75f759032ec258465fbdde073484c78eec1756a9e5","observation_id":"982d628b-f3e8-4a9b-af47-f524d73fcd9d","resolution":{"observed_at":"2026-08-12T04:44:26.435311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.415452Z","title":"SpeechT5: Unified-modal encoder-decoder pre-training for spoken language processing,","venue":null,"work_id":"afcab632-5b3a-449c-9345-d68aec979db8","year":2022},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.939645Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:07173b5af69191b50da67a607ebf89fb4aa1bda8b3f42f48d153125b2de40cfb","observation_id":"a2d75a61-cb1c-437a-bc2b-b126ea13604c","resolution":{"observed_at":"2026-08-12T04:44:26.421186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.397900Z","title":"SpeechLM: Enhanced speech pre-training with unpaired textual data,","venue":null,"work_id":"240401af-8f79-4f10-b91d-883b367a1b92","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.943271Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:d30a28f58d0254005f07ff31de273d5641fd5328a558a8d71a9056e22af0307a","observation_id":"85514468-a283-4725-a254-7017d248630d","resolution":{"observed_at":"2026-08-12T04:44:26.404239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.05187","last_updated":"2023-12-08T17:18:42Z","snapshot_observed_at":"2026-08-19T22:26:55.755677Z","submitted_at":"2023-12-08T17:18:42Z","title":"Seamless: Multilingual Expressive and Streaming Speech Translation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.05187","snapshot_observed_at":"2026-08-12T04:44:25.946785Z","title":"Seamless: Multi- lingual expressive and streaming speech translation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.946785Z"},"links":{"cited_paper":"/paper/2312.05187","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:678c1c1d3817644503a4959e5cbb468dba33af5e1f771f18e9ac8ef19777e8ce","observation_id":"86f05e72-8f1e-4a42-a720-c8b46a45e795","resolution":{"observed_at":"2026-08-12T04:44:25.946785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.383589Z","title":"M-adapter: Modality adaptation for end-to-end speech-to-text translation,","venue":null,"work_id":"099a83af-0147-416f-8dfa-8d7398c1d326","year":2022},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.950774Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:30b0e16e1016a36c565bf1036fa84b697791e3f1222a9beb0a3142c218b991e2","observation_id":"e74b7024-a6db-44f6-98ec-f26174eec041","resolution":{"observed_at":"2026-08-12T04:44:26.387882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.371210Z","title":"MAESTRO: Matched speech text representa- tions through modality matching,","venue":null,"work_id":"8a51301d-0ff6-49cf-b1bd-9ec26b818362","year":2022},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.954853Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:eb1f04bbe414a4e9e7ba5cb424f5998b72bcc18e2c7b3ccea53f3f3c412aa313","observation_id":"c0d740d2-39c1-4905-b8be-c793fb4494d9","resolution":{"observed_at":"2026-08-12T04:44:26.375306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.357098Z","title":"Cjst: Ctc compressor based joint speech and text training for decoder-only asr,","venue":null,"work_id":"95b20e22-5a98-4ac9-a84e-0bf0343c828e","year":null},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.959632Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:b7ba8a1358f3365630c09fe10ae4c062b59932089468867c6b560c60646ad2c3","observation_id":"c00317ff-d02d-48b0-9d02-c6ad41888208","resolution":{"observed_at":"2026-08-12T04:44:26.361980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.341033Z","title":"Lora: Low-rank adaptation of large language models,","venue":null,"work_id":"16afcfc1-12b3-47d2-9a3c-ecc3f994d218","year":2022},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.969315Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:4bcf8d0159d68e6f9fb43cb66859995effb6900d95c274768ddc2adeb16e9029","observation_id":"3b3ba2a4-aa92-49ee-829b-85887d30e4fd","resolution":{"observed_at":"2026-08-12T04:44:26.346175Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07607","last_updated":"2024-12-31T21:11:07Z","snapshot_observed_at":"2026-08-19T12:52:01.179757Z","submitted_at":"2024-11-12T07:30:29Z","title":"CJST: CTC Compressor based Joint Speech and Text Training for Decoder-Only ASR","version":2},"cited_work":{"arxiv_id":"2411.07607","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.07607","snapshot_observed_at":"2026-08-12T04:44:26.021330Z","title":"CJST: CTC Compressor based Joint Speech and Text Training for Decoder-Only ASR","venue":"eess.AS","work_id":"fa867cc8-0505-4fe4-b7f4-e4f6d0c587f3","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.964450Z"},"links":{"cited_paper":"/paper/2411.07607","citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:4aa5da6cec3043eef173a5a47f21f8f1cd1b07d1f9a0df0763796cef47172fe5","observation_id":"7a096d0b-5c63-45e5-9e99-4ca0b139bb8c","resolution":{"observed_at":"2026-08-12T04:44:26.028455Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.299442Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":"27372d74-f27d-4d0a-bf3c-df02804fbed2","year":2020},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.977083Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:4526e5ea71a57532e8097ff7f9b8ec179202c1b3d5cda8ad6edd268c45717748","observation_id":"8c2d6d40-f604-424b-abfb-076846be80d0","resolution":{"observed_at":"2026-08-12T04:44:26.310290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.326882Z","title":"A CTC alignment-based non- autoregressive transformer for end-to-end automatic speech recognition,","venue":null,"work_id":"de625fc8-835a-47c0-accb-2c0b6ff252b3","year":2023},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.973181Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:570fbcf7c473df90eabf67dc0a4c375cf3fb7973c8c8ddf4398fba52d478b6e7","observation_id":"d6c1a48e-e5b1-4d07-9ecf-d18ee2837067","resolution":{"observed_at":"2026-08-12T04:44:26.332237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.276784Z","title":"Unsu- pervised cross-lingual representation learning at scale,","venue":null,"work_id":"afbc1fda-0c9e-44bd-ba49-5c6138cd3949","year":2020},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.985859Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:5be3d3e6a13bd9e12431a93dfad043b24025a431ef30a590d67e786d237e57d3","observation_id":"490ff960-f90d-44d6-a8f0-fde17f19def0","resolution":{"observed_at":"2026-08-12T04:44:26.281353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:25.980965Z","title":"Zero: Memory optimizations toward training trillion parameter models,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.980965Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:39ad1e0a0d5fc814c2d443687bb243decff33908c01b3711db80b8624448ae4b","observation_id":"4e3c3e89-b628-42c0-b2c0-3cfcd5f1b980","resolution":{"observed_at":"2026-08-12T04:44:25.980965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:44:26.532437Z","title":"4552–4572","venue":null,"work_id":"c8490dea-e8ed-493f-957c-a56c1f19113d","year":2024},"citing_paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T04:44:25.886366Z"},"links":{"citing_paper":"/paper/2412.01145"},"observation_digest":"sha256:f50d9a1b7e204cff9bb7e1ac4f78cbfd27eb28c979a9f7a98c16a35ca69db0e5","observation_id":"7a11f7ff-e3ee-4275-81e7-b331e95a2782","resolution":{"observed_at":"2026-08-12T04:44:26.537792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.01145","last_updated":"2025-07-03T22:18:03Z","latest_version":2,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-18T05:22:14.276691Z","submitted_at":"2024-12-02T05:42:33Z","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":22,"verified_exact":0,"verified_fuzzy":25},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 2 inbound Pith citation observations for arXiv:2412.01145."}