{"as_of":"2026-08-10T16:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:71bfd5ba0b2906002e6b1ea6da47a0c3ebd93585bc1cc1869673a1bb82adb95a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":25,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T16:15:39.513319Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T21:17:24.595834Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2410.10781","last_updated":"2025-03-02T14:37:53Z","snapshot_observed_at":"2026-08-02T20:46:05.666977Z","submitted_at":"2024-10-14T17:50:28Z","title":"When Attention Sink Emerges in Language Models: An Empirical View","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-16T17:41:03.674759Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2410.10781"},"observation_digest":"sha256:fdebaffac1f47521a932e37f5e1ae52b06635c3be02b8ca27dd162c4d6ad34ac","observation_id":"4c4a8870-83d9-4669-b8e3-f76ba36b05fd","resolution":{"observed_at":"2026-05-16T17:41:03.735502Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-10T16:15:39.513319Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.13428","last_updated":"2026-06-01T01:21:19Z","snapshot_observed_at":"2026-08-10T15:55:46.722217Z","submitted_at":"2025-01-23T07:21:08Z","title":"Softplus Attention with Re-weighting Boosts Length Extrapolation in Large Language Models","version":6},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-10T16:15:39.513319Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2501.13428"},"observation_digest":"sha256:fad9d92cc429805c45a3300c9a8fc154bd978142cf0253515b2f19c45a2b0f5c","observation_id":"696ec69e-2f2b-4cef-a77d-81ec2fb280fa","resolution":{"observed_at":"2026-08-10T16:15:39.513319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-10T13:42:23.796832Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16215","last_updated":"2025-07-18T21:37:05Z","snapshot_observed_at":"2026-08-10T13:55:08.174224Z","submitted_at":"2025-01-27T17:07:20Z","title":"Smarter Together: Combining Large Language Models and Small Models for Physiological Signals Visual Inspection","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T13:42:23.796832Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2501.16215"},"observation_digest":"sha256:8f9353930e20499ba1ffdedd750ce8f7d1de73efc8d3f617d8433b08a3c724fe","observation_id":"8dc96183-e38f-4ce2-9f52-f5f2eb4a0572","resolution":{"observed_at":"2026-08-10T13:42:23.796832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-10T13:41:14.511574Z","title":"Theory, analysis, and best practices for sigmoid self-attention, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.16271","last_updated":"2025-01-27T18:05:28Z","snapshot_observed_at":"2026-08-10T13:54:38.063522Z","submitted_at":"2025-01-27T18:05:28Z","title":"From Molecules to Mixtures: Learning Representations of Olfactory Mixture Similarity using Inductive Biases","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-10T13:41:14.511574Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2501.16271"},"observation_digest":"sha256:3e56cce915166b0f3f181d698652405b2cf412b4208f5c9af78d059bb3610fde","observation_id":"8d03266f-9e1e-4f01-b669-d63b7e13e42e","resolution":{"observed_at":"2026-08-10T13:41:14.511574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-10T00:04:10.926525Z","title":"Ramapuram, F","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.18322","last_updated":"2026-06-17T19:55:52Z","snapshot_observed_at":"2026-08-09T23:51:33.951035Z","submitted_at":"2025-01-30T13:04:54Z","title":"A Unified Perspective on the Dynamics of Deep Transformers","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-10T00:04:10.926525Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2501.18322"},"observation_digest":"sha256:0f1f83d2925d160219dc3a0bb83e6a6c2c4a3fef5bd59af8d7ac097fa8a63798","observation_id":"31572126-1d04-47c5-a89a-8b8c22a77a88","resolution":{"observed_at":"2026-08-10T00:04:10.926525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-09T19:40:10.082924Z","title":"Ramapuram, F","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00281","last_updated":"2025-05-24T02:54:56Z","snapshot_observed_at":"2026-08-10T13:54:38.541789Z","submitted_at":"2025-02-01T02:36:14Z","title":"Sigmoid Self-Attention has Lower Sample Complexity than Softmax Self-Attention: A Mixture-of-Experts Perspective","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-09T19:40:10.082924Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2502.00281"},"observation_digest":"sha256:46e79b6cccc612fd30cd65a934ca1822c748c7348538ce6aafe3ab9e2d0c86dd","observation_id":"83fe0c0b-b786-43e1-a462-f18671deb5dd","resolution":{"observed_at":"2026-08-09T19:40:10.082924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-09T13:05:31.676260Z","title":"Theory, analysis, and best practices for sigmoid self-attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02209","last_updated":"2025-02-04T10:46:39Z","snapshot_observed_at":"2026-08-09T17:03:08.069652Z","submitted_at":"2025-02-04T10:46:39Z","title":"On the Expressivity of Selective State-Space Layers: A Multivariate Polynomial Approach","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T13:05:31.676260Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2502.02209"},"observation_digest":"sha256:0565cd71d6ccbe5ae3d99db90fa047ac81abc1ee8607c87ba88c9d42741394d2","observation_id":"54681b64-712b-4e5c-83a1-834e2d45ba3d","resolution":{"observed_at":"2026-08-09T13:05:31.676260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-08T15:37:37.515845Z","title":"Theory, analysis, and best practices for sigmoid self-attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06415","last_updated":"2025-02-26T01:59:40Z","snapshot_observed_at":"2026-08-08T15:29:19.875570Z","submitted_at":"2025-02-10T12:54:17Z","title":"Systematic Outliers in Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T15:37:37.515845Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2502.06415"},"observation_digest":"sha256:487df34d0b05aef491b53f72527c92b89eccca6794279220b2809cf30336c4b6","observation_id":"1fbe8f8a-bc61-4074-905e-185e3b70a6d6","resolution":{"observed_at":"2026-08-08T15:37:37.515845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T11:46:59.381114Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01562","last_updated":"2025-06-02T11:38:10Z","snapshot_observed_at":"2026-08-09T17:03:07.147265Z","submitted_at":"2025-06-02T11:38:10Z","title":"Unpacking Softmax: How Temperature Drives Representation Collapse, Compression, and Generalization","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:46:59.381114Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.01562"},"observation_digest":"sha256:2cc508011a57cd2083ad5ce663f24ed1616e2a11ca777daacb3892ba80b2a290","observation_id":"c125d649-43ce-4369-910f-d547675c7ca4","resolution":{"observed_at":"2026-08-07T11:46:59.381114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T06:10:19.778678Z","title":"Theory, Analysis , and Best Practices for Sigmoid Self - Attention , September 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-10T05:55:56.471654Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.778678Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:fa7fca67684c72ce76c18e4a564fd75e6f14f0e097d39485230574f11bac0d33","observation_id":"2737a320-d6ae-4494-b67f-3d9d3bf4cc1d","resolution":{"observed_at":"2026-08-07T06:10:19.778678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-06T20:00:43.782864Z","title":"Theory, analysis, and best practices for sigmoid self-attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04239","last_updated":"2025-07-06T04:15:34Z","snapshot_observed_at":"2026-08-09T23:49:22.098595Z","submitted_at":"2025-07-06T04:15:34Z","title":"Scaling Context Requires Rethinking Attention","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T20:00:43.782864Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2507.04239"},"observation_digest":"sha256:4f316b2c4a70cfe1f339221aad42d985d078c12e62701d326ff00d28f2367fee","observation_id":"ffc2af75-1f8e-4578-b29a-6f39eeb387ac","resolution":{"observed_at":"2026-08-06T20:00:43.782864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-06T13:39:10.426250Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.20453","last_updated":"2025-09-06T03:33:17Z","snapshot_observed_at":"2026-08-09T17:02:54.638838Z","submitted_at":"2025-07-28T01:07:22Z","title":"Your Attention Matters: to Improve Model Robustness to Noise and Spurious Correlations","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T13:39:10.426250Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2507.20453"},"observation_digest":"sha256:da37834fba1a45d98480c84702f6b699636a529ccec82fbec4be8988f39c8c6c","observation_id":"3b93fb8e-4c74-4dcb-aca6-d41795f98465","resolution":{"observed_at":"2026-08-06T13:39:10.426250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2604.17324","last_updated":"2026-06-07T04:11:52Z","snapshot_observed_at":"2026-08-01T08:49:51.657800Z","submitted_at":"2026-04-19T08:33:48Z","title":"Capacity-Controlled Global Attention for Graph Transformers","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T06:27:24.557866Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2604.17324"},"observation_digest":"sha256:478479e64e8cb35d00d0e92b43141033aa06aa611ef6d2c6e241034081ff3962","observation_id":"e38e9aa0-9fda-4b27-abae-cde295d70edc","resolution":{"observed_at":"2026-05-10T06:31:30.767360Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06501","last_updated":"2026-05-19T06:54:59Z","snapshot_observed_at":"2026-08-02T05:36:24.593005Z","submitted_at":"2026-05-07T16:18:55Z","title":"Cubit: Token Mixer with Kernel Ridge Regression","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-08T12:38:19.925573Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06501"},"observation_digest":"sha256:cfcd6ae2f2fb9dcb5a1894a83639958ba39f0db3ec3b1d913795425477f11c1f","observation_id":"f86867b3-92f9-4a80-a47b-29f35ab9a31c","resolution":{"observed_at":"2026-05-11T19:06:10.787346Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06501","last_updated":"2026-05-19T06:54:59Z","snapshot_observed_at":"2026-08-02T05:36:24.593005Z","submitted_at":"2026-05-07T16:18:55Z","title":"Cubit: Token Mixer with Kernel Ridge Regression","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T22:34:36.108826Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06501"},"observation_digest":"sha256:04c5b4dd2669705243aaea80282926cf0ac66498f79f104d2d991870784e1191","observation_id":"b7950fc4-5026-4447-862f-bced4a513992","resolution":{"observed_at":"2026-05-20T22:39:11.009258Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.06611","last_updated":"2026-05-07T17:28:55Z","snapshot_observed_at":"2026-08-03T12:02:38.374401Z","submitted_at":"2026-05-07T17:28:55Z","title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-08T12:11:04.146711Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.06611"},"observation_digest":"sha256:cd947092d79a36f7acbb871fe5e235baba50a85df3c27a705318dd5e21dac091","observation_id":"a8f9fc8d-b472-4959-8fea-6820f08824cb","resolution":{"observed_at":"2026-05-11T19:21:08.516991Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.08504","last_updated":"2026-05-12T18:33:07Z","snapshot_observed_at":"2026-07-06T23:20:43.010758Z","submitted_at":"2026-05-08T21:37:27Z","title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-12T02:29:20.796512Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.08504"},"observation_digest":"sha256:2b382032d88d46757b4694abe0c528cbfd591f8cb0be6aaeb909bdd1e63074c1","observation_id":"b1fd8067-78c2-4245-94e1-f0862b4855c7","resolution":{"observed_at":"2026-05-12T07:36:40.797845Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.08504","last_updated":"2026-05-12T18:33:07Z","snapshot_observed_at":"2026-07-06T23:20:43.010758Z","submitted_at":"2026-05-08T21:37:27Z","title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-14T21:03:25.624300Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.08504"},"observation_digest":"sha256:6a8059118a4628b057a5b2f1f4f0d33a5b4561227c5087a56b97fbf3b36513da","observation_id":"2ae133e3-518a-4b55-8959-1c2719b83d1e","resolution":{"observed_at":"2026-05-14T21:19:29.111658Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.10123","last_updated":"2026-07-28T06:51:39Z","snapshot_observed_at":"2026-08-07T09:22:28.887723Z","submitted_at":"2026-05-11T07:38:52Z","title":"Complex-Valued Phase-Coherent Transformer","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-12T03:34:20.893124Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.10123"},"observation_digest":"sha256:a33ef4dfaa7bc4c5953898eeafa69a1d6b2fa43598d9ff0c9d7baa3b98a15b5d","observation_id":"d20d30e7-4356-401e-b839-9e29cde7c142","resolution":{"observed_at":"2026-05-12T07:16:28.147412Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-02T14:31:47.302932Z","title":"Code:https://github.com/apple/ml-sigmoid-attention • Saratchandran et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.10123","last_updated":"2026-07-28T06:51:39Z","snapshot_observed_at":"2026-08-07T09:22:28.887723Z","submitted_at":"2026-05-11T07:38:52Z","title":"Complex-Valued Phase-Coherent Transformer","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T14:31:47.302932Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.10123"},"observation_digest":"sha256:e269033fa2b0f29789af551a08ab4a72219152601685877f8a42e969f8a795c6","observation_id":"d057774a-4e16-460f-8f77-6f7cac95eb0f","resolution":{"observed_at":"2026-08-02T14:31:47.302932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.20798","last_updated":"2026-05-20T06:43:34Z","snapshot_observed_at":"2026-08-02T17:01:43.729235Z","submitted_at":"2026-05-20T06:43:34Z","title":"Most Transformer Modifications Still Do Not Transfer at 1-3B: A 2020-2026 Update to Narang et al. (2021) with Downstream Evaluation and a Noise Floor","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-21T06:15:47.451870Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.20798"},"observation_digest":"sha256:0c7061c451adb98bdadfd26ca76780eaff5b8457be8f3e9621d0ee0ac78e6a91","observation_id":"16d1e8f8-de0a-4812-82f2-b7527e32b3e9","resolution":{"observed_at":"2026-05-21T06:19:42.072047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2605.21070","last_updated":"2026-05-20T11:56:15Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T11:56:15Z","title":"Towards Understanding Self-Pretraining for Sequence Classification","version":1},"reference_index":178,"source":"arxiv_source","source_observed_at":"2026-05-21T05:29:58.809024Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2605.21070"},"observation_digest":"sha256:9e8db73f7372be321337d86186f929481d4b2a529fca100cd741a6634636eb39","observation_id":"830e1fcf-1c26-46f4-8683-29111e1fc631","resolution":{"observed_at":"2026-05-21T05:33:58.908338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2606.02332","last_updated":"2026-06-02T05:51:37Z","snapshot_observed_at":"2026-08-08T18:54:54.423179Z","submitted_at":"2026-06-01T14:42:06Z","title":"Forget Attention: Importance-Aware Attention Is All You Need","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.948032Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2606.02332"},"observation_digest":"sha256:6e62b1b65612084481289bc29291eaeab87439c30db10b072696c83579f8a9bc","observation_id":"9c37073b-4edb-41c7-8e1f-4e2442124a49","resolution":{"observed_at":"2026-07-01T23:06:21.024853Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":"2409.04431","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-02T21:17:24.595834Z","title":"arXiv:2409.04431 , year =","venue":null,"work_id":"bd8c2a0c-3687-469e-b571-f246b747a7a8","year":2024},"citing_paper":{"arxiv_id":"2606.08105","last_updated":"2026-06-06T11:10:54Z","snapshot_observed_at":"2026-08-06T19:43:58.625241Z","submitted_at":"2026-06-06T11:10:54Z","title":"A Unifying View of Attention Sinks: Two Algorithms, Two Solutions","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T19:51:41.329591Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2606.08105"},"observation_digest":"sha256:8c6d831aa731f6a9deeaca05bffcb092fcbd185d287fd18a2b7fa093aa38c75e","observation_id":"a2f19bb8-9c0a-4305-87cd-56ab4aabf127","resolution":{"observed_at":"2026-07-02T21:17:24.597286Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-07-11T20:06:55.102341Z","title":"Theory, analysis, and best practices for sigmoid self-attention.arXiv preprint arXiv:2409.04431,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04319","last_updated":"2026-07-05T14:08:59Z","snapshot_observed_at":"2026-08-09T19:18:42.431865Z","submitted_at":"2026-07-05T14:08:59Z","title":"Legible-by-Construction: Attention and End-to-End Transformers","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T20:06:55.102341Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2607.04319"},"observation_digest":"sha256:d6fa91df28346316b007695d4ee9b29ac4bdf4e73a64874a8552927160c93e63","observation_id":"b4ea3e01-b25e-48db-861b-8dc620cab92f","resolution":{"observed_at":"2026-07-11T20:06:55.102341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2409.04431/citation-record","integrity":"/paper/2409.04431/integrity","json":"/paper/2409.04431/citation-record.json","paper":"/paper/2409.04431"},"outbound":[],"paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T13:53:56.915968Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 25 inbound Pith citation observations for arXiv:2409.04431."}