{"as_of":"2026-08-10T12:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a7f1234bc14a3061312f35172e914726e4e25ed1b1382f8c0fc307048ce19f8c","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:28:50.475371Z","state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.18145/citation-record","integrity":"/paper/2506.18145/integrity","json":"/paper/2506.18145/citation-record.json","paper":"/paper/2506.18145"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.01771","last_updated":"2024-02-01T07:15:58Z","snapshot_observed_at":"2026-08-09T03:15:24.817751Z","submitted_at":"2024-02-01T07:15:58Z","title":"BlackMamba: Mixture of Experts for State-Space Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01771","snapshot_observed_at":"2026-08-06T23:28:43.443477Z","title":"Blackmamba: Mixture of experts for state-space models.arXiv preprint arXiv:2402.01771, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:43.443477Z"},"links":{"cited_paper":"/paper/2402.01771","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:90cd0f0b9e877b567838074733076ea818e95653c0de31ec734e44b62fe8a019","observation_id":"43287aa0-aa24-4114-8f03-90e3dd1c037d","resolution":{"observed_at":"2026-08-06T23:28:43.443477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:43.499406Z","title":"Piqa: Reasoning about phys- ical commonsense in natural language","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:43.499406Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:c50c04d62a42ae8d2b70d18171b4dd5df7c2eaede9f1486eef64857757caf992","observation_id":"83cea58f-8948-4109-a043-0e7ffc771652","resolution":{"observed_at":"2026-08-06T23:28:43.499406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:43.597598Z","title":"Xception: Deep learning with depthwise separable convolutions","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:43.597598Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:34b37b9d1ce06972d716fd94b0981505bdd2373540f32cccb0a5e3376458809f","observation_id":"905298a4-452f-4a8b-abc2-30eac0cb9f5d","resolution":{"observed_at":"2026-08-06T23:28:43.597598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-06T23:28:43.733064Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge.arXiv preprint arXiv:1803.05457, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:43.733064Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:df1ba41bfd450fef86adf9c6b7348638a3545dd715ce2efb10b9d0cf8aaac16c","observation_id":"c46e0e3e-e359-4387-a097-f2eb2b377377","resolution":{"observed_at":"2026-08-06T23:28:43.733064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07987","last_updated":"2024-09-30T21:19:29Z","snapshot_observed_at":"2026-08-03T12:04:56.535363Z","submitted_at":"2023-12-13T09:00:21Z","title":"SwitchHead: Accelerating Transformers with Mixture-of-Experts Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07987","snapshot_observed_at":"2026-08-06T23:28:43.909348Z","title":"Switchhead: Accelerating transformers with mixture-of-experts attention.ArXiv, abs/2312.07987, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:43.909348Z"},"links":{"cited_paper":"/paper/2312.07987","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:b92289f5fc5f20b23ae87fb7dc6fbeeaf4a9a8f7882fa20890c21cab7cbf2651","observation_id":"836bbf3b-d613-4e08-a16c-1ca30db11d10","resolution":{"observed_at":"2026-08-06T23:28:43.909348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21060","last_updated":"2024-05-31T17:50:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:50:01Z","title":"Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21060","snapshot_observed_at":"2026-08-06T23:28:44.089192Z","title":"Transformers are ssms: Generalized models and efficient algorithms through structured state space duality.arXiv preprint arXiv:2405.21060, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.089192Z"},"links":{"cited_paper":"/paper/2405.21060","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:9d9949b59e7c87c90e088f4abfb17c9897a4171ee4d6e5bb01937359190c83f9","observation_id":"6def8f38-b877-4575-b2c4-07c3daf96f13","resolution":{"observed_at":"2026-08-06T23:28:44.089192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:44.224746Z","title":"Language modeling with gated convolutional networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.224746Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:fa1ba497536ce4655fb7d898e41fe77a3156d7e0c4a83cdc98ee52878999f552","observation_id":"d0faa61e-050d-48ce-887e-8be57a85cf0f","resolution":{"observed_at":"2026-08-06T23:28:44.224746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19427","last_updated":"2024-02-29T18:24:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T18:24:46Z","title":"Griffin: Mixing Gated Linear Recurrences with Local Attention for Efficient Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19427","snapshot_observed_at":"2026-08-06T23:28:44.335630Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.335630Z"},"links":{"cited_paper":"/paper/2402.19427","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:f222015add5318992b45d5dd12a1387055ac34bb1dd62c36999cae5c1f10e9d2","observation_id":"e10fe4ab-9031-49e5-a69c-d6f1a50e3800","resolution":{"observed_at":"2026-08-06T23:28:44.335630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T23:28:44.437943Z","title":"Deepseek-v3 technical report.arXiv preprint arXiv: 2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.437943Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:f40f9ae11e26bea1b527fed13132f93f4d38d7dd02d724da3f99fef4fe9da384","observation_id":"b2623c6a-600b-4baa-996b-eb17d8615318","resolution":{"observed_at":"2026-08-06T23:28:44.437943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13676","last_updated":"2024-11-20T19:51:25Z","snapshot_observed_at":"2026-08-02T12:55:25.835610Z","submitted_at":"2024-11-20T19:51:25Z","title":"Hymba: A Hybrid-head Architecture for Small Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13676","snapshot_observed_at":"2026-08-06T23:28:44.573565Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.573565Z"},"links":{"cited_paper":"/paper/2411.13676","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:742834730f8ea6d7ed726ea1cbc08f3bcea7b2e1fa4ae7b7eced1c1de1275c92","observation_id":"b8137f2f-5bd4-4a54-b85c-6ec7b1077466","resolution":{"observed_at":"2026-08-06T23:28:44.573565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:59.215437Z","title":"Glam: Efficient scaling of language models with mixture-of-experts","venue":null,"work_id":"abda99e1-a525-41c0-b46a-9250bc6f411e","year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.707931Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:228876f51bbc6220fc67cbea055d610b223b5f50d007edf89ba2624b8b2e900c","observation_id":"dc7a3b5f-ddcd-4426-851f-209ecbd3d6c0","resolution":{"observed_at":"2026-08-06T23:28:59.374873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:44.793964Z","title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning.Neural networks, 107:3–11, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.793964Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:078eaa75152e239c2fb98a40772c027f87eee8df9a519b798269afd92a031d4a","observation_id":"a7df6d69-ed71-45b9-a18d-71ada9c8e15f","resolution":{"observed_at":"2026-08-06T23:28:44.793964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:44.978622Z","title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity.Journal of Machine Learning Research, 23(120):1–39, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:44.978622Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:20b5119a795c2729af5de1cd069c64d4503d21e287b1e5daef8b2c10cd89a581","observation_id":"6d6e9960-19c2-40a0-afc9-f3fa0c8f72ab","resolution":{"observed_at":"2026-08-06T23:28:44.978622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:58.614832Z","title":"MegaBlocks: Efficient Sparse Training with Mixture-of-Experts.Proceedings of Machine Learning and Systems, 5, 2023","venue":null,"work_id":"2d435302-10ee-4e97-8a24-21fec8dd7a5e","year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.079024Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:6c874edf7addbd7fc084335741cf1d8b33b33e0b6b74da7f2c04150c03289f2a","observation_id":"bb44b742-94fc-481c-a7a0-17291622754d","resolution":{"observed_at":"2026-08-06T23:28:58.791464Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00752","last_updated":"2024-05-31T17:55:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-01T18:01:34Z","title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00752","snapshot_observed_at":"2026-08-06T23:28:45.430800Z","title":"Mamba: Linear-time sequence modeling with selective state spaces","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.430800Z"},"links":{"cited_paper":"/paper/2312.00752","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:fac90e6108c24d39b703451493a426d42a8b5ac8829eb760b868d86d2d213849","observation_id":"94081f34-4cba-4873-aeee-707bd58cb256","resolution":{"observed_at":"2026-08-06T23:28:45.430800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:45.516878Z","title":"On the parameterization and initialization of diagonal state space models.Advances in Neural Information Processing Systems, 35:35971–35983, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.516878Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:4b5c8d00e1f83d6fdded77e6f6bf204a5ec17cd16e8b3ecb23b88a410eb96c1a","observation_id":"e12d6c0b-211c-4a93-bdb0-c7cf4d2c7704","resolution":{"observed_at":"2026-08-06T23:28:45.516878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.00396","last_updated":"2022-08-05T17:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-10-31T03:32:18Z","title":"Efficiently Modeling Long Sequences with Structured State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.00396","snapshot_observed_at":"2026-08-06T23:28:45.671513Z","title":"Efficiently modeling long sequences with structured state spaces.arXiv preprint arXiv:2111.00396, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.671513Z"},"links":{"cited_paper":"/paper/2111.00396","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:84b4de4649874392497785b831f28729da1935c0272c498b781b033eec129541","observation_id":"67c50fef-9c16-4056-8e91-38278f1381bf","resolution":{"observed_at":"2026-08-06T23:28:45.671513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:45.786752Z","title":"Combining recurrent, convolutional, and continuous-time models with linear state space layers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.786752Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:33b4545151969f4f800bc712cfc9d5c77b0e7dd09c01e89607d13a781f236fb4","observation_id":"c413497a-fa2a-441e-ba27-0545387fcba7","resolution":{"observed_at":"2026-08-06T23:28:45.786752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:45.897623Z","title":"Diagonal state spaces are as effective as structured state spaces.Advances in Neural Information Processing Systems, 35:22982–22994, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:45.897623Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:9bbddbe63a59c596717af1bef80757651f52de7fb841e1ba383241d191b48044","observation_id":"09768454-b360-48a0-8baf-485b82b2c5f8","resolution":{"observed_at":"2026-08-06T23:28:45.897623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:46.108175Z","title":"Deep residual learning for image recognition","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.108175Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:65c524fe8db6a765ca275059d42f5b1a2abb1a0113bbe36b591273d9eaedcdb8","observation_id":"6da4b044-d2e7-4dbf-8c3f-371556bb434d","resolution":{"observed_at":"2026-08-06T23:28:46.108175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:46.274737Z","title":"Adaptive mixtures of local experts.Neural computation, 3(1):79–87, 1991","venue":null,"work_id":null,"year":1991},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.274737Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:d1598bd7dc97732350d7df2230737084218f832895acfea2c9710e4516400615","observation_id":"dec6715b-72a4-4251-a876-fc1c35c4a298","resolution":{"observed_at":"2026-08-06T23:28:46.274737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-08T06:16:25.839566Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-06T23:28:46.353703Z","title":"Mixtral of experts.arXiv preprint arXiv:2401.04088, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.353703Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:8f2c0173fbec357ec2fee2857001310fb8c29227989c7bebfd58da721c76be12","observation_id":"4f70fb5d-8f5c-4c11-a193-7ba223988bb9","resolution":{"observed_at":"2026-08-06T23:28:46.353703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:46.505103Z","title":"A new approach to linear filtering and prediction problems","venue":null,"work_id":null,"year":1960},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.505103Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:a7cb8d02d7879f5cb6c304a566c2a38863d2d6498827bb15408cd3ca673ee5b0","observation_id":"0e642c8c-3db6-47e8-ae35-161c68f8d323","resolution":{"observed_at":"2026-08-06T23:28:46.505103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16668","last_updated":"2020-06-30T10:42:02Z","snapshot_observed_at":"2026-08-07T09:27:36.420559Z","submitted_at":"2020-06-30T10:42:02Z","title":"GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.16668","snapshot_observed_at":"2026-08-06T23:28:46.643613Z","title":"Gshard: Scaling giant models with condi- tional computation and automatic sharding.arXiv preprint arXiv:2006.16668, 2020","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.643613Z"},"links":{"cited_paper":"/paper/2006.16668","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:c4bf15827f99f3e3612b7a0fcc760eb0bd85d244fbe41ed86233b87290b1582f","observation_id":"e16b7d02-cb39-45d6-8d00-b62b76a6db75","resolution":{"observed_at":"2026-08-06T23:28:46.643613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:57.544841Z","title":"Zettlemoyer, and Lili Yu","venue":null,"work_id":"4d4e88da-c146-43c8-aa4f-0fee502f8f3b","year":2025},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.736726Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:01d778808595d162405c7bf61c899b650ef12ab097924ed8f663e5bc1d95abdc","observation_id":"ee647e48-a092-4177-a0d0-ec782870658a","resolution":{"observed_at":"2026-08-06T23:28:57.725981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19887","last_updated":"2024-07-03T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-28T23:55:06Z","title":"Jamba: A Hybrid Transformer-Mamba Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19887","snapshot_observed_at":"2026-08-06T23:28:46.846085Z","title":"Jamba: A hybrid transformer-mamba language model.ArXiv, abs/2403.19887, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.846085Z"},"links":{"cited_paper":"/paper/2403.19887","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:349c7a136e05e44d46ccb9718b958bc36461db0aa3ae15f299b96b96f885bb17","observation_id":"e9984a85-245c-4353-a756-9348a54c54bf","resolution":{"observed_at":"2026-08-06T23:28:46.846085Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08612","last_updated":"2023-10-16T16:27:02Z","snapshot_observed_at":"2026-08-01T23:16:44.787779Z","submitted_at":"2023-04-17T20:59:49Z","title":"Bridging Discrete and Backpropagation: Straight-Through and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08612","snapshot_observed_at":"2026-08-06T23:28:46.939960Z","title":"Bridging discrete and backpropagation: Straight-through and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:46.939960Z"},"links":{"cited_paper":"/paper/2304.08612","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:34ab6162856787db3783fe709f2b3689e3604d00b75e1fe62e84d83aa272cd24","observation_id":"d45837fc-9079-4fa4-ba32-e5467ec42d13","resolution":{"observed_at":"2026-08-06T23:28:46.939960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00811","last_updated":"2023-10-01T22:43:57Z","snapshot_observed_at":"2026-07-06T16:26:13.862006Z","submitted_at":"2023-10-01T22:43:57Z","title":"Sparse Backpropagation for MoE Training","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00811","snapshot_observed_at":"2026-08-06T23:28:47.044827Z","title":"Sparse backpropagation for moe training.arXiv preprint arXiv:2310.00811, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.044827Z"},"links":{"cited_paper":"/paper/2310.00811","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:ab524c3ab6513d02b01f8f7e0aed997b140d782ef736a58a178c59b2cdd56453","observation_id":"c1e42e00-6894-4642-9e8d-f855d9d5d89c","resolution":{"observed_at":"2026-08-06T23:28:47.044827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:47.219802Z","title":"Megalodon: Efficient llm pretraining and inference with unlimited context length.Advances in Neural Information Processing Systems, 37:71831–71854, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.219802Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:5d8365b3b8937e420af4b4b8a00b19eebd95270a607cc2fde80d7d057ea8e0f3","observation_id":"6606f6ea-2c7c-4f94-901b-872060968c50","resolution":{"observed_at":"2026-08-06T23:28:47.219802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.13947","last_updated":"2022-07-02T17:58:04Z","snapshot_observed_at":"2026-07-06T13:25:27.826746Z","submitted_at":"2022-06-27T01:50:18Z","title":"Long Range Language Modeling via Gated State Spaces","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.13947","snapshot_observed_at":"2026-08-06T23:28:47.318283Z","title":"Long range language modeling via gated state spaces.arXiv preprint arXiv:2206.13947, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.318283Z"},"links":{"cited_paper":"/paper/2206.13947","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:45f91d5410a660e60cfacf15a4b94c2558adbefe94e4123ad92ca8a82a231216","observation_id":"cae67504-a2d9-484c-b850-292e263f985b","resolution":{"observed_at":"2026-08-06T23:28:47.318283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08313","last_updated":"2025-01-14T18:50:05Z","snapshot_observed_at":"2026-07-06T20:21:04.675084Z","submitted_at":"2025-01-14T18:50:05Z","title":"MiniMax-01: Scaling Foundation Models with Lightning Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08313","snapshot_observed_at":"2026-08-06T23:28:47.467603Z","title":"Minimax-01: Scaling foundation models with lightning attention.arXiv preprint arXiv: 2501.08313, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.467603Z"},"links":{"cited_paper":"/paper/2501.08313","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:378788024ad604aed88e71335ffcdbc85c92468f11041db723898ea3191108a1","observation_id":"75b0fa63-25e0-4298-919c-f44cf83c6db1","resolution":{"observed_at":"2026-08-06T23:28:47.467603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07143","last_updated":"2024-08-09T22:37:25Z","snapshot_observed_at":"2026-08-02T14:27:24.212588Z","submitted_at":"2024-04-10T16:18:42Z","title":"Leave No Context Behind: Efficient Infinite Context Transformers with Infini-attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.07143","snapshot_observed_at":"2026-08-06T23:28:47.577496Z","title":"Leave no context behind: Efficient infinite context transformers with infini-attention.arXiv preprint arXiv:2404.07143, 101, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.577496Z"},"links":{"cited_paper":"/paper/2404.07143","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:9cdf752bfcf3c373d16a97a60647c0eceb5526bcb24ce03c8ff5b551d55628e6","observation_id":"d6599066-3923-4656-89b0-2d1bf0bb8266","resolution":{"observed_at":"2026-08-06T23:28:47.577496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:57.089537Z","title":"Gpt-4 technical report.PREPRINT, 2023","venue":null,"work_id":"f53b9691-5622-4538-9020-1b210145ede4","year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.717086Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:66784b1be16db851fa6629a4cfddcc79770612ac51f12197925bcc4a75df9cd8","observation_id":"54c02f4e-3540-46df-8266-6405fd93c791","resolution":{"observed_at":"2026-08-06T23:28:57.234742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06031","last_updated":"2016-06-20T09:37:17Z","snapshot_observed_at":"2026-08-05T16:27:22.031013Z","submitted_at":"2016-06-20T09:37:17Z","title":"The LAMBADA dataset: Word prediction requiring a broad discourse context","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06031","snapshot_observed_at":"2026-08-06T23:28:47.881059Z","title":"The lambada dataset: Word prediction requiring a broad discourse context.arXiv preprint arXiv:1606.06031, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:47.881059Z"},"links":{"cited_paper":"/paper/1606.06031","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:518dbfc7385bb71df8a544a1449e3d1f18adf83abbe56d460bd67639da884f5f","observation_id":"9b0522f5-e80f-4e78-b3ed-040980ecea0a","resolution":{"observed_at":"2026-08-06T23:28:47.881059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04248","last_updated":"2024-04-25T17:01:52Z","snapshot_observed_at":"2026-08-09T03:26:38.274258Z","submitted_at":"2024-02-06T18:56:35Z","title":"Can Mamba Learn How to Learn? A Comparative Study on In-Context Learning Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04248","snapshot_observed_at":"2026-08-06T23:28:48.103876Z","title":"Can mamba learn how to learn? a comparative study on in-context learning tasks.arXiv preprint arXiv:2402.04248, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.103876Z"},"links":{"cited_paper":"/paper/2402.04248","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:01dc08a1aa721129848d1793d0a916c4a4a7e9f7a04f886a46049214d738d21f","observation_id":"4f650784-fa38-4c4c-a27f-d99778d3d4de","resolution":{"observed_at":"2026-08-06T23:28:48.103876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04081","last_updated":"2024-02-26T17:04:41Z","snapshot_observed_at":"2026-08-09T10:57:15.553549Z","submitted_at":"2024-01-08T18:35:07Z","title":"MoE-Mamba: Efficient Selective State Space Models with Mixture of Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04081","snapshot_observed_at":"2026-08-06T23:28:48.243629Z","title":"Moe- mamba: Efficient selective state space models with mixture of experts.ArXiv, abs/2401.04081, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.243629Z"},"links":{"cited_paper":"/paper/2401.04081","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:6a3771d3e1216a38f62924a39cdabefd3b1a13621c0c8f27320e39a7dfa48d3f","observation_id":"bd727df7-b8f5-44f9-902a-6bcf81b59c89","resolution":{"observed_at":"2026-08-06T23:28:48.243629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:48.340135Z","title":"Hyena hierarchy: Towards larger convolutional language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.340135Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:2eab25c2ec06fc93614fe88e81df8e2bcdb323a0e8b69d2ef306cbf0beef247e","observation_id":"fb089896-4dd5-4405-8b45-bad5ce2d1b4a","resolution":{"observed_at":"2026-08-06T23:28:48.340135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07522","last_updated":"2025-02-28T02:20:49Z","snapshot_observed_at":"2026-08-06T19:04:16.718797Z","submitted_at":"2024-06-11T17:50:51Z","title":"Samba: Simple Hybrid State Space Models for Efficient Unlimited Context Language Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07522","snapshot_observed_at":"2026-08-06T23:28:48.482116Z","title":"Samba: Simple hybrid state space models for efficient unlimited context language modeling.arXiv preprint arXiv:2406.07522, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.482116Z"},"links":{"cited_paper":"/paper/2406.07522","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:96076d3d7f1d54c7fcdefa81bb239a9197b7a769565c63e24c4ce8d3bf846322","observation_id":"30f7755b-eb4d-44ce-b16b-f869de9c78ce","resolution":{"observed_at":"2026-08-06T23:28:48.482116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:48.627523Z","title":"Winogrande: An adversarial winograd schema challenge at scale.Communications of the ACM, 64(9):99–106, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.627523Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:9691b172e1e6c4ec99bb1a3221eb0635358c02a41fff755ab1f173e39d33a183","observation_id":"1c796a33-950a-43c5-acdb-f00f3e1f381c","resolution":{"observed_at":"2026-08-06T23:28:48.627523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.05202","last_updated":"2020-02-12T19:57:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-02-12T19:57:13Z","title":"GLU Variants Improve Transformer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.05202","snapshot_observed_at":"2026-08-06T23:28:48.818437Z","title":"Glu variants improve transformer.arXiv preprint arXiv:2002.05202, 2020","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:48.818437Z"},"links":{"cited_paper":"/paper/2002.05202","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:a256162afe30b5c0ee1ab5e1c263bd0f087b64fd13f77c2266007eb5c9aee1a0","observation_id":"24412b46-7042-4eee-8a49-9b6c0a21a618","resolution":{"observed_at":"2026-08-06T23:28:48.818437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-06T23:28:49.087728Z","title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer.arXiv preprint arXiv:1701.06538, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.087728Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:f2c3a65755442757fb793ff1ce433d57e2d9efe935ed5834e7c6274a93d17fe3","observation_id":"c44c032e-0940-4f05-a05d-79c3be739b1c","resolution":{"observed_at":"2026-08-06T23:28:49.087728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:56.324746Z","title":"SlimPajama: A 627B token cleaned and deduplicated version of RedPajama, 2023","venue":null,"work_id":"e5cedfc7-ee5b-4e0f-900e-46bcc89820d6","year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.216391Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:64d39788e11748e11eeb6b716fd118f33da421a8fbcbc417150c1bec0034da2c","observation_id":"792221a1-f14e-465a-96db-a1573809227b","resolution":{"observed_at":"2026-08-06T23:28:56.485189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:49.370138Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.370138Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:10e7357c0cdf6388822aa68023ef1c9243c179ef250d5a642b8da476ca790143","observation_id":"864a5499-7074-4f6b-b132-448e424ed5e4","resolution":{"observed_at":"2026-08-06T23:28:49.370138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:55.775178Z","title":"Selective structured state-spaces for long-form video understanding","venue":null,"work_id":"a2271f9e-63cc-4f26-90fc-76fce65b94b7","year":2023},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.574830Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:3280d48151dde0827672e86d8619685f3e789358016c9c04f42568a628a33408","observation_id":"d70f3bf8-a48c-48fc-9180-6ceeea3f8d1c","resolution":{"observed_at":"2026-08-06T23:28:55.944852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:49.695127Z","title":"On layer normalization in the transformer architecture","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.695127Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:93d777352219fa424fea3b968be8c8047531401ba391a7569867103e59b3e542","observation_id":"3de527c2-e87a-4fd2-b81b-cbfde9ae1558","resolution":{"observed_at":"2026-08-06T23:28:49.695127Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06464","last_updated":"2025-03-06T06:57:34Z","snapshot_observed_at":"2026-08-03T00:27:18.402796Z","submitted_at":"2024-12-09T13:09:04Z","title":"Gated Delta Networks: Improving Mamba2 with Delta Rule","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06464","snapshot_observed_at":"2026-08-06T23:28:49.824833Z","title":"Gated delta networks: Improving mamba2 with delta rule.arXiv preprint arXiv:2412.06464, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:49.824833Z"},"links":{"cited_paper":"/paper/2412.06464","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:1fdc12545c6b482991ececbbfc4ae9d627fc5784fcad27c6f951a527abe4e437","observation_id":"d84efa12-7a6e-4cad-9b5f-1d0037426333","resolution":{"observed_at":"2026-08-06T23:28:49.824833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.07830","last_updated":"2019-05-19T23:57:23Z","snapshot_observed_at":"2026-07-31T00:09:56.948833Z","submitted_at":"2019-05-19T23:57:23Z","title":"HellaSwag: Can a Machine Really Finish Your Sentence?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1905.07830","snapshot_observed_at":"2026-08-06T23:28:50.055659Z","title":"Hellaswag: Can a machine really finish your sentence?arXiv preprint arXiv:1905.07830, 2019","venue":null,"work_id":null,"year":1905},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:50.055659Z"},"links":{"cited_paper":"/paper/1905.07830","citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:f94c60582868cc11ca342a69bcd18b829ef7ecdc2ab08ba18322bd096f3b69a2","observation_id":"5d383351-55f5-4721-b6e5-e56ccdb39648","resolution":{"observed_at":"2026-08-06T23:28:50.055659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:50.224917Z","title":"Root mean square layer normalization.Advances in Neural Information Processing Systems, 32, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:50.224917Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:be6144f82398ef6c7d27de90d060b61d3d87f23a61e2a77d48573bbeedbf0350","observation_id":"2e27d699-5201-495b-a7e8-785a0d232894","resolution":{"observed_at":"2026-08-06T23:28:50.224917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T23:28:54.864748Z","title":"Mixture of attention heads: Selecting attention heads per token","venue":null,"work_id":"a4bc8d5b-b6a7-44c1-84ac-1f37d7178100","year":2022},"citing_paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T23:28:50.475371Z"},"links":{"citing_paper":"/paper/2506.18145"},"observation_digest":"sha256:85934eba3d4be9134e60476ee9de2c38a53b4c34e68a83e32f4a195c0192ed22","observation_id":"dcb95f6e-96bd-4b32-9146-075f88b1ac5a","resolution":{"observed_at":"2026-08-06T23:28:55.074817Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.18145","last_updated":"2025-06-22T19:26:55Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T03:15:44.813103Z","submitted_at":"2025-06-22T19:26:55Z","title":"Routing Mamba: Scaling State Space Models with Mixture-of-Experts Projection"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":42,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 0 inbound Pith citation observations for arXiv:2506.18145."}