{"as_of":"2026-08-09T17:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3811714b520cc3d6599f90a02669230b8645b80937f38b1bfb9d26c32b4374c8","coverage":[{"denominator":45,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T11:23:41.055567Z","state":"measured"},{"denominator":55,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":55,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T22:21:10.661789Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T20:30:07.740339Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-08-06T22:21:10.661789Z","title":"Peri-ln: Revisiting normalization layer in the transformer architecture","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.22049","last_updated":"2025-07-03T16:54:09Z","snapshot_observed_at":"2026-08-07T23:11:38.262545Z","submitted_at":"2025-06-27T09:45:15Z","title":"GPAS: Accelerating Convergence of LLM Pretraining via Gradient-Preserving Activation Scaling","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T22:21:10.661789Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2506.22049"},"observation_digest":"sha256:53d4415d36e3428a69422dd5d89b40173ba4b921ea558ed8bb66852f6c08be6c","observation_id":"5ed40db9-0ac8-49d2-9536-daa3fb9bbafe","resolution":{"observed_at":"2026-08-06T22:21:10.661789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-08-03T06:37:52.586372Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2601.22580","last_updated":"2026-06-04T11:50:53Z","snapshot_observed_at":"2026-08-03T06:37:49.104766Z","submitted_at":"2026-01-30T05:21:57Z","title":"SpanNorm: Reconciling Training Stability and Performance in Deep Transformers","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T06:37:52.586372Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2601.22580"},"observation_digest":"sha256:11143dcfe299eb731b4e666e5346a8bc85effab2270c54cc5395c06c832016bb","observation_id":"06bcf884-d9c7-4e6d-947d-a84f125f271c","resolution":{"observed_at":"2026-08-03T06:37:52.586372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-08-03T04:44:55.564583Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.04396","last_updated":"2026-06-17T18:22:54Z","snapshot_observed_at":"2026-08-08T18:55:31.308399Z","submitted_at":"2026-02-04T10:25:24Z","title":"LoRDO: Distributed Low-Rank Optimization with Infrequent Communication","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T04:44:55.564583Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2602.04396"},"observation_digest":"sha256:8ee811fe8501eac825b5dca3b08e227f37cd9eb6baf90ce47f784567bd1664d2","observation_id":"b04c715f-4d2d-4480-a55f-e04f410edc69","resolution":{"observed_at":"2026-08-03T04:44:55.564583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2602.08064","last_updated":"2026-05-21T09:52:18Z","snapshot_observed_at":"2026-08-04T01:56:43.373244Z","submitted_at":"2026-02-08T17:17:56Z","title":"SiameseNorm: Breaking the Barrier to Reconciling Pre/Post-Norm","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T11:11:51.440058Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2602.08064"},"observation_digest":"sha256:def6198b93c8a2d5f323afa8da724106595cb8013db225ed584ca99bedb6a5b1","observation_id":"a43618bd-4db6-4598-b4fb-d7af19f0727e","resolution":{"observed_at":"2026-05-22T11:14:47.951414Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2604.15259","last_updated":"2026-04-22T15:48:32Z","snapshot_observed_at":"2026-07-06T23:02:49.302337Z","submitted_at":"2026-04-16T17:35:49Z","title":"Stability and Generalization in Looped Transformers","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T12:21:16.015362Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2604.15259"},"observation_digest":"sha256:8311474071a12c1d04530251fd93b3bce0cf5311072c23646f20274c5876a6ff","observation_id":"78dcf9be-8b7a-4367-933a-18d8627f13ed","resolution":{"observed_at":"2026-05-10T12:25:22.716182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2604.23434","last_updated":"2026-04-25T20:12:21Z","snapshot_observed_at":"2026-07-06T23:09:38.799345Z","submitted_at":"2026-04-25T20:12:21Z","title":"When Does Removing LayerNorm Help? Activation Bounding as a Regime-Dependent Implicit Regularizer","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-08T08:29:58.495518Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2604.23434"},"observation_digest":"sha256:555d16e120f5d3f3e7697886e7ed28672d33c43a81d34ab06d59bf5c3a2653ca","observation_id":"9a9dcb7a-7994-4f6a-85c5-7ebfefe6c14a","resolution":{"observed_at":"2026-05-11T20:36:10.321368Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2606.05957","last_updated":"2026-06-04T09:54:08Z","snapshot_observed_at":"2026-08-08T05:42:33.934596Z","submitted_at":"2026-06-04T09:54:08Z","title":"Dead Directions: Geometric Singular Learning","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-28T03:20:09.365073Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2606.05957"},"observation_digest":"sha256:a77aa79e76e0bd1dcc2c08d2cb5a320bd62456fd8f2b7d892993ba66095c5c07","observation_id":"c0dc39b1-a0ee-4ea3-b7d2-21ab322a31f0","resolution":{"observed_at":"2026-07-02T11:36:55.173661Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2606.19491","last_updated":"2026-06-17T18:28:37Z","snapshot_observed_at":"2026-07-31T17:54:27.925523Z","submitted_at":"2026-06-17T18:28:37Z","title":"Algebraic Dead Directions in LayerNorm Transformers: A Forward-Pass-Only Diagnostic at LLM Scale","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-26T21:14:11.815522Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2606.19491"},"observation_digest":"sha256:b3239ed9c3e3b485265561e5405e3908b433dc879eec725d6ac7a837921688cd","observation_id":"e0605a6f-b5f6-43ad-b366-a132f84ce771","resolution":{"observed_at":"2026-07-04T00:29:15.314353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":"2502.02732","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-07-04T20:30:07.740339Z","title":"Peri- LN : Revisiting normalization layer in the transformer architecture","venue":null,"work_id":"1a9ea840-10bb-4724-b3da-d867345370a7","year":2025},"citing_paper":{"arxiv_id":"2606.25971","last_updated":"2026-07-17T13:42:08Z","snapshot_observed_at":"2026-08-02T10:13:56.529128Z","submitted_at":"2026-06-24T15:40:26Z","title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-06-25T20:05:09.179627Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2606.25971"},"observation_digest":"sha256:7a2dcb290f5ac87f47c6909d81b760d429c4a7f278cce3ea2e1876a23f55966b","observation_id":"04607fde-e876-48eb-a945-022b466d8c50","resolution":{"observed_at":"2026-07-04T20:30:07.742267Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02732","snapshot_observed_at":"2026-08-02T10:14:07.666821Z","title":"Peri-ln: Revisiting normalization layer in the transformer architecture","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.25971","last_updated":"2026-07-17T13:42:08Z","snapshot_observed_at":"2026-08-02T10:13:56.529128Z","submitted_at":"2026-06-24T15:40:26Z","title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-02T10:14:07.666821Z"},"links":{"cited_paper":"/paper/2502.02732","citing_paper":"/paper/2606.25971"},"observation_digest":"sha256:9a0d215678151d82eb4ac5febb734ba6460e4924331a3283ee5feaf8923780bf","observation_id":"688d377c-aa05-4460-9e65-c8e76ea8e7f5","resolution":{"observed_at":"2026-08-02T10:14:07.666821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.02732/citation-record","integrity":"/paper/2502.02732/integrity","json":"/paper/2502.02732/citation-record.json","paper":"/paper/2502.02732"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.851628Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.851628Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:4a60eb1892940f497f6aa57c636183eed13ca2e9334f7647fcdc1809423ac39b","observation_id":"f434b86a-30db-4c69-9e5d-f50cb7796c1b","resolution":{"observed_at":"2026-08-09T11:23:40.851628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1607.06450","last_updated":"2016-07-21T19:57:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-07-21T19:57:52Z","title":"Layer Normalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1607.06450","snapshot_observed_at":"2026-08-09T11:23:40.877867Z","title":"J., Kiros, J","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.877867Z"},"links":{"cited_paper":"/paper/1607.06450","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:7b967054a0cca7ef8b961a39f15d079dcaccd454c8d03014ad2bdc572861962c","observation_id":"33036fcf-68a6-40d5-9d13-e66ae0bcde08","resolution":{"observed_at":"2026-08-09T11:23:40.877867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.892330Z","title":"Piqa: Reasoning about physical commonsense in natural language","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.892330Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:513378ebac0f91becddab296afbdeebc6fd5c34cc20f1687308ad5e3beebd04c","observation_id":"489b6201-372d-49a7-abd8-e83b6d28aec4","resolution":{"observed_at":"2026-08-09T11:23:40.892330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.569784Z","title":"L., and Simonyan, K","venue":null,"work_id":"62040f19-b8ef-4ced-aa59-ee3ffbee581d","year":2021},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.895994Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:20528fe0b3f6277b8074b13672a2c771e7ace20517de9bc918773fa5624afd91","observation_id":"6fa722ab-77cf-46aa-9809-f9ffe06d2493","resolution":{"observed_at":"2026-08-09T11:23:41.573228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.558920Z","title":"L., and Simonyan, K","venue":null,"work_id":"1b32ce3a-6aa0-4fab-bb19-951517118c99","year":2021},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.899516Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:57ad2d7ba14762df0ab0d9b719bb4c4a69ecd2008c0e9f927b84d8872c3d1373","observation_id":"ae05f490-bfe2-4729-8b53-be2f80d62ad1","resolution":{"observed_at":"2026-08-09T11:23:41.562745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.546942Z","title":"M., Thorne, J., and Yun, S","venue":null,"work_id":"99ae1bf4-3a55-407c-bd8b-2311513f4500","year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.902568Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:a326e1b52f15f17589c78bfa56ddf8fe231edc41dba98b3da80498bba37535fe","observation_id":"1b316606-3123-4869-a6db-25d10e720cde","resolution":{"observed_at":"2026-08-09T11:23:41.550491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-09T11:23:40.905850Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge, 2018","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.905850Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:96610e3b591c35cb976499c86cdd87a8abc28bc100c5413b7530ad973b06fbe8","observation_id":"a70a568f-fcf9-44cd-abf6-3f6c792b270e","resolution":{"observed_at":"2026-08-09T11:23:40.905850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16039","last_updated":"2024-10-13T04:46:00Z","snapshot_observed_at":"2026-07-06T18:19:42.946629Z","submitted_at":"2024-05-25T03:24:32Z","title":"MoEUT: Mixture-of-Experts Universal Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16039","snapshot_observed_at":"2026-08-09T11:23:40.909701Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.909701Z"},"links":{"cited_paper":"/paper/2405.16039","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:c9e363bef7083e4e0576258735c4bbc8c4f082bdf486a2f39a700613d0cffa53","observation_id":"1a540c13-cf09-4887-992e-30b8a00c4bbb","resolution":{"observed_at":"2026-08-09T11:23:40.909701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.536408Z","title":"and Smith, S","venue":null,"work_id":"2e3cd1a8-37c9-49a3-8b99-41e01818b8a9","year":2020},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.914052Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:d9bcadcfef9299867a9047bdbe525becb9a4fbb06e07dc31b44e1e48044d6e82","observation_id":"daa6c897-9273-45f5-a1e6-dad8f48b6d34","resolution":{"observed_at":"2026-08-09T11:23:41.540017Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.524971Z","title":"P., Caron, M., Geirhos, R., Alabdulmohsin, I., Jenatton, R., Beyer, L., Tschannen, M., Arnab, A., Wang, X., Ruiz, C","venue":null,"work_id":"80103416-823c-49f4-b2cb-abc0dd5ccc02","year":2023},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.918117Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:5efd1b379eaf9aeaabc39faf25c9e760b64e403436072d4a8b3fc0230d2f801d","observation_id":"8229a91f-b51e-4e77-bebb-c2f276e0c3d5","resolution":{"observed_at":"2026-08-09T11:23:41.528861Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.513655Z","title":"Gpt3.int8(): 8-bit matrix multiplication for transformers at scale","venue":null,"work_id":"f3de8de8-5077-4fda-ad68-d3713b6050d8","year":2022},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.922084Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:a9305af483593b8f355b92b0f6914182c38b76840a999cc7d5a72608466b6afd","observation_id":"4ff7c646-81ee-4f4f-a085-e65cd0e673bd","resolution":{"observed_at":"2026-08-09T11:23:41.518073Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-09T11:23:40.925710Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.925710Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:b5adaaf6c3cd2b2a2cf39cce878a808530a2d3f1641108afad8af6f9be2a3b30","observation_id":"90beec48-6507-4b43-85cb-26e22324facb","resolution":{"observed_at":"2026-08-09T11:23:40.925710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12517","last_updated":"2025-02-10T09:37:59Z","snapshot_observed_at":"2026-08-07T19:54:09.137276Z","submitted_at":"2024-09-19T07:15:58Z","title":"Scaling FP8 training to trillion-token LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12517","snapshot_observed_at":"2026-08-09T11:23:40.929720Z","title":"Scaling FP8 training to trillion-token llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.929720Z"},"links":{"cited_paper":"/paper/2409.12517","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:0b326e1451d9e80ebd2916983f1b0acc5064d6c16533edd40d77d667bc5e060d","observation_id":"f804b551-985d-46b4-9dc7-9f50336ff2f0","resolution":{"observed_at":"2026-08-09T11:23:40.929720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.933794Z","title":"A framework for few-shot language model evaluation, 07 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.933794Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:03d7727146599e7408ab8e2a9aae66e4b2fbbb7530265676afe5ead27e0d9a26","observation_id":"682aae9f-fafc-44c9-9a0d-e5c5c0bda128","resolution":{"observed_at":"2026-08-09T11:23:40.933794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-09T11:23:40.937603Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.937603Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:ad855db57b2889a4497f0600d6b27a02c437fac12c3bf96420389969979807f9","observation_id":"d09e8149-6d69-44e0-bad1-c5ce28e1ffdb","resolution":{"observed_at":"2026-08-09T11:23:40.937603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.941936Z","title":"Identity mappings in deep residual networks","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.941936Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:2bf4d6a98ffaa89a708167b29228125011409c3554b0d3f9783148181e233100","observation_id":"41f1e2a8-b304-4d18-bc08-be09ac3352ef","resolution":{"observed_at":"2026-08-09T11:23:40.941936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-07-06T12:54:11.616335Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-09T11:23:40.945771Z","title":"A., Welbl, J., Clark, A., Hennigan, T., Noland, E., Millican, K., van den Driessche, G., Damoc, B., Guy, A., Osindero, S., Simonyan, K., Elsen, E., Rae, J","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.945771Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:a7c6e4707e135b6dc3dde752a26ef04149ba0aac8aff1dc2947e5e4cc542c874","observation_id":"3797cbe1-dd16-44b3-ad2e-ab9d53d4116e","resolution":{"observed_at":"2026-08-09T11:23:40.945771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.502872Z","title":"A., Khyalia, S., Jung, J., Goka, H., and Lee, H","venue":null,"work_id":"7a947a71-4ce4-470f-b867-6b73746579c4","year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.949975Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:ad52facf2a52c16568cc87b01bb3e9cd21f163a466da78117b94798de8bd0f33","observation_id":"f1bb609d-33cc-42c1-8579-99c6a4319361","resolution":{"observed_at":"2026-08-09T11:23:41.506469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11794","last_updated":"2025-04-21T17:48:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T17:42:57Z","title":"DataComp-LM: In search of the next generation of training sets for language models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11794","snapshot_observed_at":"2026-08-09T11:23:40.953362Z","title":"Y., Bansal, H., Guha, E","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.953362Z"},"links":{"cited_paper":"/paper/2406.11794","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:f576b0b66f14927bd2645196a03ee5adf710d722af847b3fb8ce1c8a4b27c673","observation_id":"7c0e1cdc-9a3b-4385-864c-ac3734dfee30","resolution":{"observed_at":"2026-08-09T11:23:40.953362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13795","last_updated":"2025-08-04T10:49:54Z","snapshot_observed_at":"2026-08-05T13:37:28.213031Z","submitted_at":"2024-12-18T12:39:53Z","title":"Mix-LN: Unleashing the Power of Deeper Layers by Combining Pre-LN and Post-LN","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13795","snapshot_observed_at":"2026-08-09T11:23:40.957150Z","title":"Mix-ln: Unleashing the power of deeper layers by combining pre-ln and post-ln","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.957150Z"},"links":{"cited_paper":"/paper/2412.13795","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:0ed6a2dad6c281184173976a9a3a6b00b08035264f61be7d222fab5255bc410c","observation_id":"a31f75b4-e492-4d28-a9f1-391160482c09","resolution":{"observed_at":"2026-08-09T11:23:40.957150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01131","last_updated":"2025-04-23T18:41:43Z","snapshot_observed_at":"2026-08-03T18:00:51.343975Z","submitted_at":"2024-10-01T23:50:09Z","title":"nGPT: Normalized Transformer with Representation Learning on the Hypersphere","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01131","snapshot_observed_at":"2026-08-09T11:23:40.961082Z","title":"ngpt: Normalized transformer with representation learning on the hypersphere","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.961082Z"},"links":{"cited_paper":"/paper/2410.01131","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:7a535c011891184e7faa24a00cb41d766fd92886ed23e801b6b06d6b67b437c9","observation_id":"387a7cb2-6a81-4870-b072-6f18762890be","resolution":{"observed_at":"2026-08-09T11:23:40.961082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.965013Z","title":"Pointer sentinel mixture models, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.965013Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:54d75b2767663b51c8ed1d4d4812339323c0461b38d9d8c496ce926d19b7ac05","observation_id":"c45c6c18-8285-42b1-a488-f82ce8b71647","resolution":{"observed_at":"2026-08-09T11:23:40.965013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.968440Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.968440Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:0cd2772c3688ff003ad7602ee5c2e51ecc3da5be8844f1253e48d6e4e14f3875","observation_id":"1d8a9af5-cecf-4a19-90e4-fe0ef20036a3","resolution":{"observed_at":"2026-08-09T11:23:40.968440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00656","last_updated":"2025-10-08T07:50:45Z","snapshot_observed_at":"2026-08-08T06:58:44.493777Z","submitted_at":"2024-12-31T21:55:10Z","title":"2 OLMo 2 Furious","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00656","snapshot_observed_at":"2026-08-09T11:23:40.972128Z","title":"2 olmo 2 furious","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.972128Z"},"links":{"cited_paper":"/paper/2501.00656","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:b7e29c67e5314f90e543d0ac65a08bfbbb1d7e6cd1a54bbf2d991eb70c7fab92","observation_id":"18b61c7c-d3ba-4f10-9f51-c300bec588ee","resolution":{"observed_at":"2026-08-09T11:23:40.972128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.479735Z","title":"L., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A., Schulman, J., Hilton, J., Kelton, F., Miller, L., Simens, M., Askell, A., Welinder, P., Christiano, P","venue":null,"work_id":"7617d4cc-79be-4623-8b71-41f41e0d6c63","year":2022},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.976010Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:0daa28544bfcb6003430a324e895e4acafcf22dc5360c4a14c447a0308a2c74b","observation_id":"20b3394c-4b23-4505-8dbc-1851da4d4e73","resolution":{"observed_at":"2026-08-09T11:23:41.483607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.979249Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.979249Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:077b0216b8713a635120982bccefe421ba616ea6e5a2e7db7a0b931983354f0d","observation_id":"85ec0cc4-3662-442c-97cf-fd1d460adbfd","resolution":{"observed_at":"2026-08-09T11:23:40.979249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-09T11:23:40.982032Z","title":"G., Hardin, C., Bhupatiraju, S., Hussenot, L., Mesnard, T., Shahriari, B., Ram \\' e , A., Ferret, J., Liu, P., Tafti, P., Friesen, A., Casbon, M., Ramos, S., Kumar, R., Lan, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.982032Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:62082c73e88a52daa0578475776ff78e058548bcbf8022f8d55e2b9805cc6e4e","observation_id":"8b1e79a5-d6b3-4751-9ea3-df53f9195c07","resolution":{"observed_at":"2026-08-09T11:23:40.982032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.985508Z","title":"L., Bhagavatula, C., and Choi, Y","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.985508Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:1d41f3e79684511bf667849f25970520aa82aa992c8bc42a9230cdf348262ba3","observation_id":"cf1532cd-400e-44f5-a65b-adc92efedd93","resolution":{"observed_at":"2026-08-09T11:23:40.985508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:40.989356Z","title":"L., and Choi, Y","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.989356Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:15d4da470f3f00caa1e4841ef86a51dbaaaa09a688e08a5e0e38c9dde92ebd8d","observation_id":"5abc72c6-91bb-45f4-81cb-d482a5edef7b","resolution":{"observed_at":"2026-08-09T11:23:40.989356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17762","last_updated":"2024-08-14T16:00:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-27T18:55:17Z","title":"Massive Activations in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17762","snapshot_observed_at":"2026-08-09T11:23:40.993209Z","title":"Z., and Liu, Z","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.993209Z"},"links":{"cited_paper":"/paper/2402.17762","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:a547e4e220e44bd20fc62c52b22873dfdcea40d349cc7f74d1ef3d86dd73cbd3","observation_id":"94b3edb7-d041-4212-bc15-c788e160fcef","resolution":{"observed_at":"2026-08-09T11:23:40.993209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.16903","last_updated":"2025-07-25T05:09:17Z","snapshot_observed_at":"2026-07-06T17:09:10.513197Z","submitted_at":"2023-12-28T08:53:27Z","title":"Spike No More: Stabilizing the Pre-training of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.16903","snapshot_observed_at":"2026-08-09T11:23:40.997471Z","title":"Spike no more: Stabilizing the pre-training of large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:40.997471Z"},"links":{"cited_paper":"/paper/2312.16903","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:4f317aec0d62c4c641fed0566fa8a318b45575dc6acf9d375efbf64ca3e4746f","observation_id":"8a60c67d-ea8b-4711-8474-e9476ac3d168","resolution":{"observed_at":"2026-08-09T11:23:40.997471Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.001809Z","title":"C ommonsense QA : A question answering challenge targeting commonsense knowledge","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.001809Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:bd74174722dbcd43428fa72551d1ab36de9f85a4af658bf3914b194109bb12dc","observation_id":"d3a15d2e-cd47-41d2-9885-18a4cf15b8f9","resolution":{"observed_at":"2026-08-09T11:23:41.001809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.005504Z","title":"N., Kaiser, L., and Polosukhin, I","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.005504Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:ec27526639227bd4d312593f77668e70e8a9c9b1278ee279a95ef0ccb80c2562","observation_id":"1a0ffefa-e06a-4d59-bcc2-249464d75863","resolution":{"observed_at":"2026-08-09T11:23:41.005504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.009462Z","title":"GLUE : A multi-task benchmark and analysis platform for natural language understanding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.009462Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:b81b0844ae9224d503f0cfe3aadc17dbcb3f2804d54b775be3a46a2cfb17eab5","observation_id":"fb1782f0-9542-4ac3-97ac-ff1c35bec19b","resolution":{"observed_at":"2026-08-09T11:23:41.009462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.013426Z","title":"L., Gugger, S., Drame, M., Lhoest, Q., and Rush, A","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.013426Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:57a8d4ffbece357ebabd61dc550d07ed72c5727c18146ccda328b528a5f9b0e2","observation_id":"4f061457-a891-4704-a02e-6304a0928a86","resolution":{"observed_at":"2026-08-09T11:23:41.013426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.451630Z","title":"J., Xiao, L., Everett, K","venue":null,"work_id":"d9b57534-fb0e-4ee4-aee4-f8b5c79786fb","year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.017054Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:50465fb47cbf7516945a58a9b12fa330c0eadfd3cf12d214ff4abbc46730205f","observation_id":"bdf134be-3901-4f65-9e91-2c71f1564e43","resolution":{"observed_at":"2026-08-09T11:23:41.454954Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.440700Z","title":"On layer normalization in the transformer architecture","venue":null,"work_id":"194a3691-1d70-48a2-9c44-7793b5d16498","year":2020},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.020942Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:0828790ce85c9e14d7c350399dbce29b1425bef0b717b59c01cd8ae4b7fe5c4b","observation_id":"235e7f01-f60e-464a-9ec5-ef9bdb045830","resolution":{"observed_at":"2026-08-09T11:23:41.444567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.429250Z","title":"and Hu, E","venue":null,"work_id":"44112a3c-703b-4f35-9e97-c5b79cae2359","year":2021},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.024962Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:939db774d7d309426a4337b7f369e732ce9c3dd277a67d925edee9dcb31362de","observation_id":"23483fe6-e21b-45e7-8bee-6191beb2f6ce","resolution":{"observed_at":"2026-08-09T11:23:41.433266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03466","last_updated":"2022-03-28T08:12:14Z","snapshot_observed_at":"2026-08-07T09:04:57.100575Z","submitted_at":"2022-03-07T15:37:35Z","title":"Tensor Programs V: Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.03466","snapshot_observed_at":"2026-08-09T11:23:41.028667Z","title":"J., Babuschkin, I., Sidor, S., Liu, X., Farhi, D., Ryder, N., Pachocki, J., Chen, W., and Gao, J","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.028667Z"},"links":{"cited_paper":"/paper/2203.03466","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:973ba81682248bb24e76df0443fd94f224a8297aaa1415410f1e46fb8d8154ba","observation_id":"f5e1b34c-27a7-47fb-828b-3d361a3956a8","resolution":{"observed_at":"2026-08-09T11:23:41.028667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.418912Z","title":"Tensor programs VI: feature learning in infinite depth neural networks","venue":null,"work_id":"4951a945-0cfd-4a6d-9372-d3cd52dd6786","year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.032836Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:5abbd1961fddeb0f48f861a92c335aff0b13710ff08a0d19dce042622b94c114","observation_id":"c2c44df5-cfa7-4c95-8f64-380ab4e7127c","resolution":{"observed_at":"2026-08-09T11:23:41.422449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07191","last_updated":"2025-07-07T17:42:19Z","snapshot_observed_at":"2026-08-06T02:24:32.331127Z","submitted_at":"2024-11-11T18:05:48Z","title":"The Super Weight in Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07191","snapshot_observed_at":"2026-08-09T11:23:41.036654Z","title":"The super weight in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.036654Z"},"links":{"cited_paper":"/paper/2411.07191","citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:d0d3295490b2a29671ab358586dc18725b27ace98ca78c1535ad66b4bf931a95","observation_id":"72684422-acc7-4d9c-b5e4-255bd4e51eb3","resolution":{"observed_at":"2026-08-09T11:23:41.036654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.043808Z","title":"H ella S wag: Can a machine really finish your sentence? In Korhonen, A., Traum, D., and M \\`a rquez, L","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.043808Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:5dac96d94489c48fcfab0b7546633a6a60e89658be85aecaa63be0ec347869a9","observation_id":"daa9108c-0e48-4ae2-8efb-73377e0be55d","resolution":{"observed_at":"2026-08-09T11:23:41.043808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.408091Z","title":null,"venue":null,"work_id":"81bef0fa-ddc5-4349-9039-a8f901ef79a3","year":2023},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.047603Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:fe025b49096c2aafe60d921cde794b86a7255d89136d5d3f05b0d369c7d7ffa0","observation_id":"7ae72bcf-7902-42ec-973e-de252bd75d5f","resolution":{"observed_at":"2026-08-09T11:23:41.411931Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.397153Z","title":"and Sennrich, R","venue":null,"work_id":"aa9cebb4-cb0d-43ef-89a7-fca6e166ee75","year":2019},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.052010Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:64648f9d1c2fd8a9405535499341ff529707ea555f1272dc9392af16f4c493fe","observation_id":"15357f38-1a24-48bf-a6b7-06ac4b227d9a","resolution":{"observed_at":"2026-08-09T11:23:41.400476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T11:23:41.384192Z","title":"LIMA: less is more for alignment","venue":null,"work_id":"29399ab9-71d0-427a-bdb9-76641ad4f226","year":2023},"citing_paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture","version":3},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-09T11:23:41.055567Z"},"links":{"citing_paper":"/paper/2502.02732"},"observation_digest":"sha256:295d341f72cde71461655c5f783d9964ac1b04a2b3075d47dccfe2969f5a389c","observation_id":"71f6c8eb-862d-46d5-83e1-99b8114cfe6b","resolution":{"observed_at":"2026-08-09T11:23:41.389771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.02732","last_updated":"2025-06-06T11:19:11Z","latest_version":3,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T11:16:06.304371Z","submitted_at":"2025-02-04T21:29:47Z","title":"Peri-LN: Revisiting Normalization Layer in the Transformer Architecture"},"reference_resolution":{"displayed":45,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":31,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":45},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 45 of 45 outbound references and 10 inbound Pith citation observations for arXiv:2502.02732."}