{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4WFUSRLDMPV4UDI3AZYOR7V4MT","short_pith_number":"pith:4WFUSRLD","schema_version":"1.0","canonical_sha256":"e58b49456363ebca0d1b0670e8febc64fe6186ee2ed1aa899e00d9f2bb7e4aaf","source":{"kind":"arxiv","id":"2505.17060","version":1},"attestation_state":"computed","paper":{"title":"SALMONN-omni: A Standalone Speech LLM without Codec Injection for Full-duplex Conversation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chao Zhang, Guangzhi Sun, Jun Zhang, Lu Lu, Siyin Wang, Wenyi Yu, Xianzhao Chen, Xiaohai Tian, Xiaoyu Yang, Yuxuan Wang","submitted_at":"2025-05-17T08:13:59Z","abstract_excerpt":"In order to enable fluid and natural human-machine speech interaction, existing full-duplex conversational systems often adopt modular architectures with auxiliary components such as voice activity detectors, interrupters, conversation state predictors, or multiple LLMs. These systems, however, suffer from error accumulation across modules and struggle with key challenges such as context-dependent barge-in and echo cancellation. Recent approaches, most notably Moshi, simplify the pipeline by injecting audio codecs into the token space of a single LLM. However, such methods still incur signific"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17060","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-17T08:13:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"05460de800e92f4631d15069dee08fe963de0b60415a5c6190b2b9e00c972ad4","abstract_canon_sha256":"51d0507ab8724516887e1db70ac7d604a2c2e88b54d38084943c2d9a6719f0d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:46.401604Z","signature_b64":"tTZEGggPZb2mOZPlkvoDc5YWX/yBV88c3aq+miIWXja0/i9CEz1G3q61FGsM0yQ4e0v4LZKYGErIDUO8pgfuAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e58b49456363ebca0d1b0670e8febc64fe6186ee2ed1aa899e00d9f2bb7e4aaf","last_reissued_at":"2026-07-05T11:07:46.401077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:46.401077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SALMONN-omni: A Standalone Speech LLM without Codec Injection for Full-duplex Conversation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chao Zhang, Guangzhi Sun, Jun Zhang, Lu Lu, Siyin Wang, Wenyi Yu, Xianzhao Chen, Xiaohai Tian, Xiaoyu Yang, Yuxuan Wang","submitted_at":"2025-05-17T08:13:59Z","abstract_excerpt":"In order to enable fluid and natural human-machine speech interaction, existing full-duplex conversational systems often adopt modular architectures with auxiliary components such as voice activity detectors, interrupters, conversation state predictors, or multiple LLMs. These systems, however, suffer from error accumulation across modules and struggle with key challenges such as context-dependent barge-in and echo cancellation. Recent approaches, most notably Moshi, simplify the pipeline by injecting audio codecs into the token space of a single LLM. However, such methods still incur signific"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17060","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17060/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17060","created_at":"2026-07-05T11:07:46.401143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17060v1","created_at":"2026-07-05T11:07:46.401143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17060","created_at":"2026-07-05T11:07:46.401143+00:00"},{"alias_kind":"pith_short_12","alias_value":"4WFUSRLDMPV4","created_at":"2026-07-05T11:07:46.401143+00:00"},{"alias_kind":"pith_short_16","alias_value":"4WFUSRLDMPV4UDI3","created_at":"2026-07-05T11:07:46.401143+00:00"},{"alias_kind":"pith_short_8","alias_value":"4WFUSRLD","created_at":"2026-07-05T11:07:46.401143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13450","citing_title":"Endpoint Anticipation for Low-Latency Spoken Dialogue","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09186","citing_title":"DuplexOmni: Real-Time Listening, Seeing, Thinking, and Speaking for Full-Duplex Interaction","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00247","citing_title":"Adaptive Perturbation Selection for Contrastive Audio Decoding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20414","citing_title":"PlanRAG-Audio: Planning and Retrieval Augmented Generation for Long-form Audio Understanding","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17837","citing_title":"The Silent Thought: Modeling Internal Cognition in Full-Duplex Spoken Dialogue Models via Latent Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26388","citing_title":"Game-Time: Evaluating Temporal Dynamics in Spoken Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2603.17837","citing_title":"The Silent Thought: Modeling Internal Cognition in Full-Duplex Spoken Dialogue Models via Latent Reasoning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10199","citing_title":"How Should LLMs Listen While Speaking? A Study of User-Stream Routing in Full-Duplex Spoken Dialogue","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT","json":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT.json","graph_json":"https://pith.science/api/pith-number/4WFUSRLDMPV4UDI3AZYOR7V4MT/graph.json","events_json":"https://pith.science/api/pith-number/4WFUSRLDMPV4UDI3AZYOR7V4MT/events.json","paper":"https://pith.science/paper/4WFUSRLD"},"agent_actions":{"view_html":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT","download_json":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT.json","view_paper":"https://pith.science/paper/4WFUSRLD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17060&json=true","fetch_graph":"https://pith.science/api/pith-number/4WFUSRLDMPV4UDI3AZYOR7V4MT/graph.json","fetch_events":"https://pith.science/api/pith-number/4WFUSRLDMPV4UDI3AZYOR7V4MT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT/action/storage_attestation","attest_author":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT/action/author_attestation","sign_citation":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT/action/citation_signature","submit_replication":"https://pith.science/pith/4WFUSRLDMPV4UDI3AZYOR7V4MT/action/replication_record"}},"created_at":"2026-07-05T11:07:46.401143+00:00","updated_at":"2026-07-05T11:07:46.401143+00:00"}