{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:HRAYVZNUIGPYYFHKYIAUHEN3KA","short_pith_number":"pith:HRAYVZNU","schema_version":"1.0","canonical_sha256":"3c418ae5b4419f8c14eac2014391bb50340e84fd37b30d640c718657105d8da5","source":{"kind":"arxiv","id":"2608.09119","version":1},"attestation_state":"computed","paper":{"title":"Motif 3: Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Beomgyu Kim, Bokki Ryu, Changjin Kang, Dahye Choi, Dongpin Oh, Dongseok Kim, Gihun Cho, Hanbin Jung, Hongjoo Lee, Hyeyeon Cho, Hyukjin Kweon, Jaeheui Her, Jangwoong Kim, Jeesoo Lee, Jeongdoo Lee, Joon Son Chung, Junghwan Lim, Junhyeok Lee, Minjae Kim, Minsu Ha, Sangho Kang, Sungmin Lee, Taehyun Kim, Taewhan Kim, Wai Ting Cheung, Yeongjae Park, Youngrok Kim","submitted_at":"2026-08-10T04:53:05Z","abstract_excerpt":"We introduce Motif 3, a decoder-only Mixture-of-Experts language model with 314 billion total parameters and 13.2 billion activated per token. Each sparse MoE layer contains 384 routed experts, with eight selected per token. This fine-grained sparsity provides substantial expert capacity while limiting computation. Motif 3 is built around Grouped Differential Latent Attention (GDLA), which integrates grouped differential attention with the compressed key-value representation of Multi-head Latent Attention. The architecture further incorporates modified manifold-constrained hyper-connections, E"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.09119","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-08-10T04:53:05Z","cross_cats_sorted":[],"title_canon_sha256":"78143a66986d52cf778e73c6e69553e30bdc61f6cdebf67f12f29523faba227a","abstract_canon_sha256":"03f8185c7b52667906d398faa6436ca017f24a6eb35cb8df51f799cbcedc093b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T02:21:50.861384Z","signature_b64":"7W07kRk5DOW+9XDa+peqoc8NmPn/FCF4gNho7I+l39z8ssGTWI1H1yg2klkNj4EaOEr6xGESKEM81bn7/nYbCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c418ae5b4419f8c14eac2014391bb50340e84fd37b30d640c718657105d8da5","last_reissued_at":"2026-08-11T02:21:50.859811Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T02:21:50.859811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Motif 3: Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Beomgyu Kim, Bokki Ryu, Changjin Kang, Dahye Choi, Dongpin Oh, Dongseok Kim, Gihun Cho, Hanbin Jung, Hongjoo Lee, Hyeyeon Cho, Hyukjin Kweon, Jaeheui Her, Jangwoong Kim, Jeesoo Lee, Jeongdoo Lee, Joon Son Chung, Junghwan Lim, Junhyeok Lee, Minjae Kim, Minsu Ha, Sangho Kang, Sungmin Lee, Taehyun Kim, Taewhan Kim, Wai Ting Cheung, Yeongjae Park, Youngrok Kim","submitted_at":"2026-08-10T04:53:05Z","abstract_excerpt":"We introduce Motif 3, a decoder-only Mixture-of-Experts language model with 314 billion total parameters and 13.2 billion activated per token. Each sparse MoE layer contains 384 routed experts, with eight selected per token. This fine-grained sparsity provides substantial expert capacity while limiting computation. Motif 3 is built around Grouped Differential Latent Attention (GDLA), which integrates grouped differential attention with the compressed key-value representation of Multi-head Latent Attention. The architecture further incorporates modified manifold-constrained hyper-connections, E"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.09119","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.09119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.09119","created_at":"2026-08-11T02:21:50.860419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.09119v1","created_at":"2026-08-11T02:21:50.860419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.09119","created_at":"2026-08-11T02:21:50.860419+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRAYVZNUIGPY","created_at":"2026-08-11T02:21:50.860419+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRAYVZNUIGPYYFHK","created_at":"2026-08-11T02:21:50.860419+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRAYVZNU","created_at":"2026-08-11T02:21:50.860419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA","json":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA.json","graph_json":"https://pith.science/api/pith-number/HRAYVZNUIGPYYFHKYIAUHEN3KA/graph.json","events_json":"https://pith.science/api/pith-number/HRAYVZNUIGPYYFHKYIAUHEN3KA/events.json","paper":"https://pith.science/paper/HRAYVZNU"},"agent_actions":{"view_html":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA","download_json":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA.json","view_paper":"https://pith.science/paper/HRAYVZNU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.09119&json=true","fetch_graph":"https://pith.science/api/pith-number/HRAYVZNUIGPYYFHKYIAUHEN3KA/graph.json","fetch_events":"https://pith.science/api/pith-number/HRAYVZNUIGPYYFHKYIAUHEN3KA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA/action/storage_attestation","attest_author":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA/action/author_attestation","sign_citation":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA/action/citation_signature","submit_replication":"https://pith.science/pith/HRAYVZNUIGPYYFHKYIAUHEN3KA/action/replication_record"}},"created_at":"2026-08-11T02:21:50.860419+00:00","updated_at":"2026-08-11T02:21:50.860419+00:00"}