{"as_of":"2026-08-10T03:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8c9e11eb31e8fead935dd629edb31812125b451109632b1b680563f0d868cfce","coverage":[{"denominator":27,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":27,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T19:58:14.675856Z","state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T17:30:14.955831Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T17:30:15.054198Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"cited_work":{"arxiv_id":"2502.01659","doi":null,"metadata_source":"pith","pith_arxiv_id":"2502.01659","snapshot_observed_at":"2026-08-06T17:30:15.054198Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","venue":"cs.LG","work_id":"c39bb327-b93d-444f-8f57-cf9bbd4c08d0","year":2025},"citing_paper":{"arxiv_id":"2507.10855","last_updated":"2025-07-14T23:03:24Z","snapshot_observed_at":"2026-08-08T10:53:46.330154Z","submitted_at":"2025-07-14T23:03:24Z","title":"Sparse Fine-Tuning of Transformers for Generative Tasks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T17:30:14.955831Z"},"links":{"cited_paper":"/paper/2502.01659","citing_paper":"/paper/2507.10855"},"observation_digest":"sha256:2d55d5806894618da7f24f2fba8b0663e8203613a5606ca5ddd03fb135746f5c","observation_id":"68793c4a-d8bc-43c8-b927-4c223eb6b933","resolution":{"observed_at":"2026-08-06T17:30:15.058609Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2502.01659/citation-record","integrity":"/paper/2502.01659/integrity","json":"/paper/2502.01659/citation-record.json","paper":"/paper/2502.01659"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-09T19:58:14.552267Z","title":"A survey of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.552267Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:e4e3d8d802f90627556d7d0e83902a57e7edfe19d0055069c8b02b934118e140","observation_id":"ad0472da-339c-4524-80d8-8a306225c2c6","resolution":{"observed_at":"2026-08-09T19:58:14.552267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.100358Z","title":"Enh ancing molecular design efﬁciency: Uniting language models and ge nerative networks with genetic algorithms,","venue":null,"work_id":"0601582f-4699-416d-94f0-e1497fd19090","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.558119Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:31899a8b3162eb9202ccd772e92a60725a974f4833523f6144779204426e0685","observation_id":"b4660cd1-02b2-46f5-910e-2475dec1acb2","resolution":{"observed_at":"2026-08-09T19:58:15.105507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.086412Z","title":"Path-bigbird: An ai-driven transformer appro ach to classi- ﬁcation of cancer pathology reports,","venue":null,"work_id":"3e6298b5-2002-469b-bd98-f430e0cb4714","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.562776Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:478513c1296dfcb6f0c7875851a9f6f443cb3379f76ae834089fa98e5d34524a","observation_id":"a873b736-35bf-4859-ae49-63ca1ac51876","resolution":{"observed_at":"2026-08-09T19:58:15.091091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.071742Z","title":"Hyenadna: Long-range genomic sequence modeling at single nucleotide resolution,","venue":null,"work_id":"2d74d52c-5d15-40c4-a79d-3a23c77b6912","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.567627Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:8bd47b3c4042e00d8f7cab710242e46e9b1b52472ca0f801abf5dc0e64e1cfff","observation_id":"55288ae4-3069-4aea-bbf5-72de2c81f4fc","resolution":{"observed_at":"2026-08-09T19:58:15.076670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.056803Z","title":"Attention is all you need,","venue":null,"work_id":"6536b697-6ded-482f-b427-e2f3bfac7db7","year":2017},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.572624Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:27240ded060783b35844b68bf78a0e5c619340981174d8ac98fcac2c1f88302a","observation_id":"482daa7e-b360-45fe-83ce-ef2c1586d498","resolution":{"observed_at":"2026-08-09T19:58:15.061485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.041862Z","title":"Big bird: Transformers for longer sequences,","venue":null,"work_id":"9015bef4-63db-4780-a2a7-720b103c657d","year":2020},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.577071Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:bc0f71d316933565e2bf8efacd9d1350bae9745e98cd7d80c619294368aa2c33","observation_id":"fe2008fa-86f5-4635-9eb2-17a2a97781aa","resolution":{"observed_at":"2026-08-09T19:58:15.046896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.027113Z","title":"Longnet: Scaling transformers to 1,000,000,00 0 tokens,","venue":null,"work_id":"f2b26cc7-9cbf-481f-99ce-fdb6421aa845","year":null},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.582358Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:05aa0f6d78424981bb36fd3541d8f26e59a315f37ad7c18a9128a21dfef24814","observation_id":"111e59df-db05-4aa6-8cc8-4fb5eb861679","resolution":{"observed_at":"2026-08-09T19:58:15.032040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-09T19:58:14.591311Z","title":"Longformer: The l ong- document transformer,","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.591311Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:72247327ee5e5ff612cdc5716b3ff483f67e4af3ef72b7d5f8170285659bcd98","observation_id":"519b063f-f8c2-495f-b7b6-70a578983ac2","resolution":{"observed_at":"2026-08-09T19:58:14.591311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.997245Z","title":"Scaled Dot Product Attention,","venue":null,"work_id":"eacf4e55-065e-4692-a0c5-7949b6445262","year":null},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.596125Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:51ed5e3781a142700a71765ba81b3b0240a0d95e11e32a9ce9f540a9b6949ba3","observation_id":"a142f2ca-da0e-4d9e-ae0b-fc39cdfd7904","resolution":{"observed_at":"2026-08-09T19:58:15.001995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.982558Z","title":"xformers: A mod- ular and hackable transformer modelling library,","venue":null,"work_id":"f42f867f-4b8b-4ec3-bc3b-2a4333aa669f","year":2022},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.600723Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:eeacf25be0a026b41b97ac03826e678885794f24972586a3461b173ca130e340","observation_id":"0b861f90-77c0-41b4-80be-1ac1764db6ad","resolution":{"observed_at":"2026-08-09T19:58:14.987469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.04451","last_updated":"2020-02-18T16:01:18Z","snapshot_observed_at":"2026-07-06T08:50:12.690900Z","submitted_at":"2020-01-13T18:38:28Z","title":"Reformer: The Efficient Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.04451","snapshot_observed_at":"2026-08-09T19:58:14.605321Z","title":"Reformer: The ef ﬁcient trans- former,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.605321Z"},"links":{"cited_paper":"/paper/2001.04451","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:48b7ed64cb1d5ed983fbea33e9ac3e87442ab8c52109cff89eea8431eea1ad28","observation_id":"07c2b4a2-cc94-449b-b72d-7f07304adabf","resolution":{"observed_at":"2026-08-09T19:58:14.605321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.10509","last_updated":"2019-04-23T19:29:47Z","snapshot_observed_at":"2026-08-09T19:46:04.857927Z","submitted_at":"2019-04-23T19:29:47Z","title":"Generating Long Sequences with Sparse Transformers","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.10509","snapshot_observed_at":"2026-08-09T19:58:14.609918Z","title":"Genera ting long sequences with sparse transformers,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.609918Z"},"links":{"cited_paper":"/paper/1904.10509","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:ea27c9b76a27143a9fee37f990aa0d6570e639bd2fe1890d189f6bd9740484ab","observation_id":"31528233-5fe3-4462-bbfb-19c5e332320b","resolution":{"observed_at":"2026-08-09T19:58:14.609918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.966570Z","title":"Representing long-range context for graph neural network s with global attention,","venue":null,"work_id":"0242c555-48d2-4fdf-bdfd-70432dfbbed4","year":2021},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.615306Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:55356e7d2e38137ce55971295c6b988db89a7de178c42ebd39e8bca2fc5f47ae","observation_id":"e9c68474-da92-401f-8422-e672b75014f6","resolution":{"observed_at":"2026-08-09T19:58:14.971912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01889","last_updated":"2023-11-27T06:38:47Z","snapshot_observed_at":"2026-08-07T09:22:20.831075Z","submitted_at":"2023-10-03T08:44:50Z","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01889","snapshot_observed_at":"2026-08-09T19:58:14.619889Z","title":"Ring attention with b lockwise transformers for near-inﬁnite context,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.619889Z"},"links":{"cited_paper":"/paper/2310.01889","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:6c8d287d1401166e5214c16b9151b3fb6efba24982c90d9963cb14e4b185362f","observation_id":"f5b02652-6964-4fe2-9172-75aad3881ef4","resolution":{"observed_at":"2026-08-09T19:58:14.619889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.951428Z","title":"Blockwise parallel transformers for large context models,","venue":null,"work_id":"3b159c82-ee81-4d8e-8a3c-35f930a43562","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.624919Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:3077beece596b4d64b96d98f180c2ceddc74bf6669bb60ad2feba908be0ffecb","observation_id":"4f491a2e-409f-4af4-85d1-321a611e4390","resolution":{"observed_at":"2026-08-09T19:58:14.955929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-04T19:27:31.715261Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14509","snapshot_observed_at":"2026-08-09T19:58:14.629276Z","title":"Deepspeed ulysses: System optimizations for ena bling training of extreme long sequence transformer models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.629276Z"},"links":{"cited_paper":"/paper/2309.14509","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:c7cda96a21ff3f8f43db267c3837ac69b87c42d161bcd1df307d36b145cb1e03","observation_id":"d0c6718a-c934-411e-97ea-65eab7e4a81a","resolution":{"observed_at":"2026-08-09T19:58:14.629276Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-09T19:58:14.634125Z","title":"Megatron-lm: Training multi-billion parameter lan guage models using model parallelism,","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.634125Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:6a2d12c6768cc77d90a18e3eb29a9bc841b7abd585bc15d9cef0974c5e444704","observation_id":"3e7a8ee8-796c-4045-a9d0-b86c1b4377c4","resolution":{"observed_at":"2026-08-09T19:58:14.634125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14135","last_updated":"2022-06-23T17:53:32Z","snapshot_observed_at":"2026-07-06T13:14:48.753329Z","submitted_at":"2022-05-27T17:53:09Z","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.14135","snapshot_observed_at":"2026-08-09T19:58:14.638827Z","title":"Flashatt ention: Fast and memory-efﬁcient exact attention with io-awarenes s,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.638827Z"},"links":{"cited_paper":"/paper/2205.14135","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:f02abfdaf10f5cbd14b2936672917e34356cb0e2b504e6142c8e520da97a7282","observation_id":"dd7d6fc3-1fca-4267-9285-c34428813930","resolution":{"observed_at":"2026-08-09T19:58:14.638827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.936559Z","title":"Flashattention-2: Faster attention with bett er parallelism and work partitioning,","venue":null,"work_id":"273a4968-a4fe-402b-9a7a-cbbcb4023322","year":2023},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.643740Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:68ecd9a2a0dfa0b4c8a648d0d981acf9b964ff356ca84487156904e3924d1a5a","observation_id":"b7c3e28b-d0c1-4d88-a2a9-25be97bb4c89","resolution":{"observed_at":"2026-08-09T19:58:14.941376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.921358Z","title":"Flashattention-3: Fast and accurate attention with async hrony and low-precision,","venue":null,"work_id":"3c871e72-b51c-45e7-aace-4fbebdb9fdbe","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.648012Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:6bcbc8c67f667df80b17d6521c996ba0a2c3a3f30197ba5135534734364f57e4","observation_id":"5094d809-8176-4923-9b10-210a5d1406c7","resolution":{"observed_at":"2026-08-09T19:58:14.926557Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.01160","last_updated":"2023-06-01T21:33:59Z","snapshot_observed_at":"2026-07-06T15:36:52.977284Z","submitted_at":"2023-06-01T21:33:59Z","title":"Faster Causal Attention Over Large Sequences Through Sparse Flash Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.01160","snapshot_observed_at":"2026-08-09T19:58:14.652472Z","title":"Faster causal attention over large sequences through sparse ﬂash attenti on,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.652472Z"},"links":{"cited_paper":"/paper/2306.01160","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:13eac57a82afe142bfe7fa3968eb0a4174ae521f5e6b29a362de45a57820b6de","observation_id":"15a5d1cf-8792-4de6-a79b-b2838f604f5c","resolution":{"observed_at":"2026-08-09T19:58:14.652472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.15097","last_updated":"2024-09-24T12:56:13Z","snapshot_observed_at":"2026-08-04T13:19:47.421063Z","submitted_at":"2024-09-23T15:11:07Z","title":"Efficiently Dispatching Flash Attention For Partially Filled Attention Masks","version":2},"cited_work":{"arxiv_id":"2409.15097","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.15097","snapshot_observed_at":"2026-08-09T19:58:14.741799Z","title":"Efficiently Dispatching Flash Attention For Partially Filled Attention Masks","venue":"cs.LG","work_id":"6649b38b-67d4-4b36-b29f-1e03bf532d18","year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.657264Z"},"links":{"cited_paper":"/paper/2409.15097","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:e2c73572de358ff072b723eabd6a5f4c1c7c97b748a651ac4e47fb9c4347ab92","observation_id":"708e9999-3208-4809-80aa-f22aaee6f15d","resolution":{"observed_at":"2026-08-09T19:58:14.748858Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1805.02867","last_updated":"2018-07-28T06:51:27Z","snapshot_observed_at":"2026-07-06T06:37:55.065033Z","submitted_at":"2018-05-08T07:34:17Z","title":"Online normalizer calculation for softmax","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1805.02867","snapshot_observed_at":"2026-08-09T19:58:14.661834Z","title":"Online normalizer calcu lation for softmax,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.661834Z"},"links":{"cited_paper":"/paper/1805.02867","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:0635fc1e42b0b235d2051548e220b5e98550a1f707d3ff02fd2dbb2e2bcd1af0","observation_id":"fb76bb74-b7da-4650-8099-a2379f0770de","resolution":{"observed_at":"2026-08-09T19:58:14.661834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.905857Z","title":"On the power of some pram models,","venue":null,"work_id":"135c558d-3281-40fb-b8e0-d772b3366ade","year":1999},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.666325Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:2e455cbacc0af4f7b6c0ec5cd6f53c886f0b2aff84c7f41dd27f4d04f505cee2","observation_id":"f45fad95-f6b0-4952-b127-0159a1d532f5","resolution":{"observed_at":"2026-08-09T19:58:14.910917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-09T19:58:14.670570Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.670570Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:4d6845c8bbf1138a4106e4b095775899d03ec4c61678e8abad9cbb076e94d7cb","observation_id":"6b0649cd-c7c1-4705-b0cd-421d6f60bb44","resolution":{"observed_at":"2026-08-09T19:58:14.670570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:14.890522Z","title":"Algorithm 10xx: Suitesparse:graphblas: Graph algorithms in the language of sparse linear algebra,","venue":null,"work_id":"4ec26aae-a9ea-44a3-98e1-1ca5949d1537","year":2022},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.675856Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:7449dea8915d652153ea4c929965e63f8915455e8c8044e5ac95741f9a529001","observation_id":"57219432-3d10-46c5-8c68-3d6832828071","resolution":{"observed_at":"2026-08-09T19:58:14.895370Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T19:58:15.012178Z","title":"Available: https://arxiv.org/abs/2307","venue":null,"work_id":"e1a99a76-a372-499e-b793-f5fd7fab830a","year":null},"citing_paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-09T19:58:14.586742Z"},"links":{"citing_paper":"/paper/2502.01659"},"observation_digest":"sha256:2f397c32a082972f7988857a28d64e86dfe16536d74be5fc055ec27a150b2390","observation_id":"b0074eb4-cd9c-4d93-a543-bcf4e14d4178","resolution":{"observed_at":"2026-08-09T19:58:15.016939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.01659","last_updated":"2025-02-07T13:44:24Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T19:50:41.790456Z","submitted_at":"2025-01-31T22:05:00Z","title":"Longer Attention Span: Increasing Transformer Context Length with Sparse Graph Processing Techniques"},"reference_resolution":{"displayed":27,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":1,"verified_fuzzy":15},"total_outbound_references":27},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 27 of 27 outbound references and 1 inbound Pith citation observation for arXiv:2502.01659."}