{"as_of":"2026-08-11T01:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b426018a7856f138de74a507d76459ab73a756e6136288188b90298a06102628","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T18:15:32.936227Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.00669/citation-record","integrity":"/paper/2502.00669/integrity","json":"/paper/2502.00669/citation-record.json","paper":"/paper/2502.00669"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.811800Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.811800Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:c79360d1c40d85a353b883e3e3ca6ef4ea9e4868b22a1fd4aa969ef5f6956ce2","observation_id":"a1c68aee-393d-41ac-aa02-6dbec66f71ab","resolution":{"observed_at":"2026-08-09T18:15:32.811800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-09T18:15:32.816513Z","title":"L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.816513Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:f5cf5d646fe9228368261fc156c112e9da6fbb85410f41f72c403398b8ea4827","observation_id":"2adb4e64-ac68-4f96-a29f-93594ba5248d","resolution":{"observed_at":"2026-08-09T18:15:32.816513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02151","last_updated":"2025-04-17T18:55:45Z","snapshot_observed_at":"2026-08-10T11:04:18.421023Z","submitted_at":"2024-04-02T17:58:27Z","title":"Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02151","snapshot_observed_at":"2026-08-09T18:15:32.821396Z","title":"Jailbreaking leading safety-aligned llms with simple adaptive attacks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.821396Z"},"links":{"cited_paper":"/paper/2404.02151","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:398106e090f0a6c65dd87b4a7506911dd2683995aa5e5eda35f38da3361730c0","observation_id":"1ede8505-f567-4d25-8e08-42b71a4e83c4","resolution":{"observed_at":"2026-08-09T18:15:32.821396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-09T18:15:32.825084Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.825084Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:957b58392298c43a9d4bbae8251dd81bdc9f60ffb264b68bef08977c1079b18c","observation_id":"26357bcb-33a8-4dbc-bfdd-b147b3c2f91d","resolution":{"observed_at":"2026-08-09T18:15:32.825084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.321449Z","title":"Concentration inequalities","venue":null,"work_id":"3d72735e-6a4c-4d70-9ca8-1b5e73957b49","year":2003},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.828569Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:69e35874b63b4dc4362a8cee4126da6f002818e8cb1437d9fa16f955321722f4","observation_id":"e3865526-d003-4d3a-87a4-427909f12ae2","resolution":{"observed_at":"2026-08-09T18:15:33.324614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.831757Z","title":"A., Jagielski, M., Gao, I., Koh, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.831757Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:9f89606823d15cdc443324cc65c525e837cd4402b6f0af68a6f72659c00aa695","observation_id":"aff7a8f5-7412-4496-aa56-6b6b0e819e8c","resolution":{"observed_at":"2026-08-09T18:15:32.831757Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.16260","last_updated":"2025-04-19T14:43:00Z","snapshot_observed_at":"2026-08-03T08:20:12.021014Z","submitted_at":"2024-11-25T10:23:11Z","title":"Unraveling Arithmetic in Large Language Models: The Role of Algebraic Structures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.16260","snapshot_observed_at":"2026-08-09T18:15:32.835036Z","title":"and Wu, P.-Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.835036Z"},"links":{"cited_paper":"/paper/2411.16260","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:8af6ad54aec57de9825bb7328fb915d28002140df44b82b6e8d9e295d87fba82","observation_id":"0b6abb9b-6002-4b1f-bc8d-23831e701bbe","resolution":{"observed_at":"2026-08-09T18:15:32.835036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.304898Z","title":"J., and Wong, E","venue":null,"work_id":"6b4bd4f4-2b5e-4825-ad8a-a89eff4db4d4","year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.839435Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:5b61f9a369b5558ff05f50eea5af373658f3e9c0453187acf6b0f6ec48cd24b2","observation_id":"9ce9f81b-03ea-40de-822c-96b052c9ba27","resolution":{"observed_at":"2026-08-09T18:15:33.308169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.295109Z","title":"Shifting attention to relevance: Towards the predictive uncertainty quantification of free-form large language models","venue":null,"work_id":"a694f3a0-6699-470e-a393-61ed90b2b955","year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.842584Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:66a7718c04d4aa72cbb29cd123ee4b4dfcd11fb749d3889ef5c61fbd5a0f61ef","observation_id":"f7c32822-ae7f-4f98-8649-356701b933ca","resolution":{"observed_at":"2026-08-09T18:15:33.298182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.285265Z","title":null,"venue":null,"work_id":"d3724723-ccec-4d14-b653-603f04cb8fe6","year":2004},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.845653Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:d10289bbfc967e56076f86ac3902278a47e26ef811a318a80db7e705312dd20c","observation_id":"6d52bdbb-6091-4a04-9b94-c9c396c1bb85","resolution":{"observed_at":"2026-08-09T18:15:33.288683Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.276399Z","title":"Kto: Model alignment as prospect theoretic optimization","venue":null,"work_id":"9fde5fb0-69f3-4343-89a9-f27492e1f619","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.848997Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:677b81606d0fe389f5d5137e897edba928609bac69bf08efee242f50d957e0cf","observation_id":"11c3cc77-1867-4fc3-a571-02ad49232186","resolution":{"observed_at":"2026-08-09T18:15:33.279353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.267982Z","title":null,"venue":null,"work_id":"9328171a-5afd-4f90-b95d-5f55818c5e0a","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.852048Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:4bbbd040b1d4b493806b18284c1ba4f8da9fd10077911cb56f4fe5166c0d43b3","observation_id":"c03d6884-bc03-4688-93e7-c340d4c6acd5","resolution":{"observed_at":"2026-08-09T18:15:33.270705Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-09T18:15:32.855160Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.855160Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:a1f6d6a1437e41ad771adfd2c5acd666959f4e1a8199956c0116e5113893875f","observation_id":"4483b9e3-68ac-4695-af91-6ab4b3630c89","resolution":{"observed_at":"2026-08-09T18:15:32.855160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11801","last_updated":"2024-10-28T17:30:58Z","snapshot_observed_at":"2026-07-06T18:32:23.999019Z","submitted_at":"2024-06-17T17:48:13Z","title":"Safety Arithmetic: A Framework for Test-time Safety Alignment of Language Models by Steering Parameters and Activations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11801","snapshot_observed_at":"2026-08-09T18:15:32.858370Z","title":"Safety arithmetic: A framework for test-time safety alignment of language models by steering parameters and activations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.858370Z"},"links":{"cited_paper":"/paper/2406.11801","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:52a62af1d923e139eed35fa6d7eeda254c87285e0453e193eeeab4208a2c73b8","observation_id":"3c1bff9c-f5c7-42b4-8bd8-962cc37df406","resolution":{"observed_at":"2026-08-09T18:15:32.858370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.258754Z","title":"Catastrophic jailbreak of open-source llms via exploiting generation","venue":null,"work_id":"3d96ded6-c13d-42b0-a4e1-00c71a87e88c","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.862370Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:7c735d9aa56ee65e0921baa9d42694daa028b9017ec402dd5b077287213ecd49","observation_id":"4fd99f3d-829e-41e1-9868-6ead4daaa5cf","resolution":{"observed_at":"2026-08-09T18:15:33.261880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06120","last_updated":"2024-09-05T16:19:32Z","snapshot_observed_at":"2026-08-10T18:13:56.503000Z","submitted_at":"2024-02-09T01:10:25Z","title":"Exploring Group and Symmetry Principles in Large Language Models","version":3},"cited_work":{"arxiv_id":"2402.06120","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.06120","snapshot_observed_at":"2026-08-09T18:15:33.102260Z","title":"Exploring Group and Symmetry Principles in Large Language Models","venue":"cs.CL","work_id":"5b90e24a-7493-4198-9380-7b74e5a8420d","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.865641Z"},"links":{"cited_paper":"/paper/2402.06120","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:35595af2cac35a772980f34d9d65415fb4db640b0522f4ed93443016d977d70d","observation_id":"ed44d68c-fa37-40c4-9803-716b36f3dc31","resolution":{"observed_at":"2026-08-09T18:15:33.105794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.869252Z","title":"Neural tangent kernel: Convergence and generalization in neural networks","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.869252Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:663e740c0a6e41ae44a08b73db8f3983ec5da8e27a724397fbf77669c5c3296f","observation_id":"4031ab7e-7bf1-452d-a683-698af3fa2341","resolution":{"observed_at":"2026-08-09T18:15:32.869252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11867","last_updated":"2024-05-28T07:05:05Z","snapshot_observed_at":"2026-08-10T17:16:23.330772Z","submitted_at":"2024-02-19T06:22:09Z","title":"LoRA Training in the NTK Regime has No Spurious Local Minima","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11867","snapshot_observed_at":"2026-08-09T18:15:32.872378Z","title":"D., and Ryu, E","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.872378Z"},"links":{"cited_paper":"/paper/2402.11867","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:19ecead80384e980ea708fe9cc09be021cfb11dbdd6e4c9c70009ff4313a9587","observation_id":"d32a9366-0ad5-4618-ba3d-25249dbd4d1f","resolution":{"observed_at":"2026-08-09T18:15:32.872378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.243348Z","title":null,"venue":null,"work_id":"a46f880d-a001-43ee-85e3-d6d70a7f5ddf","year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.875935Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:098387c35435bd013a475ee8043ccea8a0c1d2b1b4dcd05651713b5d4d12bde5","observation_id":"e104d32e-07ec-4af8-b8f2-3f2f257402a6","resolution":{"observed_at":"2026-08-09T18:15:33.246489Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.233830Z","title":"Decoupling noise and toxic parameters for language model detoxification by task vector merging","venue":null,"work_id":"a96e5bc5-9754-402d-b82f-ac96fd615787","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.878996Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:5b6c2fcf487d3ef6b4ddb7755796adda714e134ec3b1de236b99493184166191","observation_id":"f7824558-fb1d-4cca-a4be-5a7281e36284","resolution":{"observed_at":"2026-08-09T18:15:33.237166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.17003","last_updated":"2025-04-07T07:23:33Z","snapshot_observed_at":"2026-08-07T07:23:22.071348Z","submitted_at":"2024-08-30T04:35:59Z","title":"Safety Layers in Aligned Large Language Models: The Key to LLM Security","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.17003","snapshot_observed_at":"2026-08-09T18:15:32.881971Z","title":"Safety layers in aligned large language models: The key to llm security","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.881971Z"},"links":{"cited_paper":"/paper/2408.17003","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:5e144a2620ecce102226e08ce18c4dfcc9be73f0838f9f2d33a6ef78e1c90ed0","observation_id":"4fb17518-3c4a-4b3e-9f8c-6a4a76fceaf7","resolution":{"observed_at":"2026-08-09T18:15:32.881971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.223689Z","title":"A kernel-based view of language model fine-tuning","venue":null,"work_id":"7923f15b-7fff-4ff8-9780-61a1bdf75b7d","year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.885303Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:bfbd05b151db7b81f5c7a9e40fd30cfb5b1a8bb2fb984228d31ee8679f2fe054","observation_id":"843d9b8d-22a8-4cee-8ba9-2514ef6e260e","resolution":{"observed_at":"2026-08-09T18:15:33.227179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.888608Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.888608Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:aa19fbd3c598ed9141c2bfa2c3a0abd1d099b337a0a18421289b996a632a6e26","observation_id":"53e4cba7-1e6b-4a40-ae76-0f260f625e4b","resolution":{"observed_at":"2026-08-09T18:15:32.888608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.207058Z","title":"Visual adversarial examples jailbreak aligned large language models","venue":null,"work_id":"a2f1f96b-31fd-4cbb-854d-dfe558ec8c22","year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.891650Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:a19fbfbfd7964b77ddcbaada008f5d2eaa2db475e74ae90156364e4813418da3","observation_id":"49939161-8884-44cd-998d-ce38929f7f38","resolution":{"observed_at":"2026-08-09T18:15:33.210298Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03693","last_updated":"2023-10-05T17:12:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-05T17:12:17Z","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03693","snapshot_observed_at":"2026-08-09T18:15:32.894727Z","title":"Fine-tuning aligned language models compromises safety, even when users do not intend to! arXiv preprint arXiv:2310.03693, 2023 b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.894727Z"},"links":{"cited_paper":"/paper/2310.03693","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:6f4f97e39b66cbce4fd23fd30e725d7a10a2ccff1ec7c7f3538b1f22c85dea08","observation_id":"c00176fb-b8bc-4b46-94f9-808b408461e6","resolution":{"observed_at":"2026-08-09T18:15:32.894727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.05946","last_updated":"2024-06-10T00:35:23Z","snapshot_observed_at":"2026-08-10T19:41:48.737592Z","submitted_at":"2024-06-10T00:35:23Z","title":"Safety Alignment Should Be Made More Than Just a Few Tokens Deep","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.05946","snapshot_observed_at":"2026-08-09T18:15:32.898587Z","title":"Safety alignment should be made more than just a few tokens deep, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.898587Z"},"links":{"cited_paper":"/paper/2406.05946","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:73dfe923a2bcc5b7ad0266be7c12faf0a78f0ce4c6aaf08676fb007b9a91fd3f","observation_id":"086a878c-dc55-4f14-b6c0-9f6361320ccb","resolution":{"observed_at":"2026-08-09T18:15:32.898587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.902109Z","title":"D., Ermon, S., and Finn, C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.902109Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:3895547fcad96aa2fbc6d7c86aa53ec18fbbfd3a33723bc35930adf85002303b","observation_id":"4169530e-8513-4cd0-8714-d4c184d8daf7","resolution":{"observed_at":"2026-08-09T18:15:32.902109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:32.904947Z","title":null,"venue":null,"work_id":null,"year":1977},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.904947Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:13dfc37efebc2cc400946f6134ff0688092c166a641bc879ae3814daac3ecfa0","observation_id":"70ae71cb-2c0f-4d63-a75f-5e0711fdf588","resolution":{"observed_at":"2026-08-09T18:15:32.904947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-09T18:15:32.907853Z","title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.907853Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:d7f2be1c79f06a6e1636e00a5c0b6b9969035375aade8c3ce7d800fb3f36d221","observation_id":"4b4c462d-e2ae-4a19-8d82-ffe5ad6a3334","resolution":{"observed_at":"2026-08-09T18:15:32.907853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-09T18:15:32.910799Z","title":"M., Hauth, A., et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.910799Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:353965f7c0dd23a14e15ecb49475730a11ef7cca00ca589fc9fcacb96d54ffde","observation_id":"7955819a-b393-4b30-a831-9d2e7c10ce98","resolution":{"observed_at":"2026-08-09T18:15:32.910799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16747","last_updated":"2024-10-22T11:53:58Z","snapshot_observed_at":"2026-07-06T18:20:14.021759Z","submitted_at":"2024-05-27T01:31:40Z","title":"Understanding Linear Probing then Fine-tuning Language Models from NTK Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16747","snapshot_observed_at":"2026-08-09T18:15:32.914089Z","title":"and Sato, I","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.914089Z"},"links":{"cited_paper":"/paper/2405.16747","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:40e975ddd28af4a79115281399982b22e9a6779e927c6b7b943daee1a91b1ed0","observation_id":"059a1817-5705-4b53-b873-b6fe5261a1b9","resolution":{"observed_at":"2026-08-09T18:15:32.914089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-09T18:15:32.917170Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.917170Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:b4aa558cc80cb0bbbc673d02b2842b1aa69055102e1218242d1f037b9e84524a","observation_id":"d4bb110e-b0be-44b3-a5d7-ee315513a6c1","resolution":{"observed_at":"2026-08-09T18:15:32.917170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-09T18:15:32.920266Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.920266Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:047165163d89a789c231b6ffe5a37b9812a20f63dff390465d4b94c5751d4803","observation_id":"260ae4dd-9bb6-4ebd-b804-7199c082a674","resolution":{"observed_at":"2026-08-09T18:15:32.920266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.184811Z","title":"On the vulnerability of safety alignment in open-access llms","venue":null,"work_id":"a6c4d958-df36-4668-b62d-895c8fc5ff9c","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.923114Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:85a44cd4627b368450cb9dae1b8f40e99a9026ad395479578c634ac8e6be140a","observation_id":"47eedbd9-d5c5-4fd1-8b79-3423b730deea","resolution":{"observed_at":"2026-08-09T18:15:33.188011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.02724","last_updated":"2025-02-02T15:57:01Z","snapshot_observed_at":"2026-08-06T13:43:29.144376Z","submitted_at":"2024-10-03T17:45:31Z","title":"Large Language Models as Markov Chains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.02724","snapshot_observed_at":"2026-08-09T18:15:32.926521Z","title":"Large language models as markov chains","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.926521Z"},"links":{"cited_paper":"/paper/2410.02724","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:393346ec5396e477b695ab43d2b4f7b84f324274f1451589a15a232cb756e874","observation_id":"051da380-7ca0-4764-b07a-2060435f37fb","resolution":{"observed_at":"2026-08-09T18:15:32.926521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.05553","last_updated":"2024-04-05T23:30:56Z","snapshot_observed_at":"2026-08-07T23:30:34.877117Z","submitted_at":"2023-11-09T17:54:59Z","title":"Removing RLHF Protections in GPT-4 via Fine-Tuning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.05553","snapshot_observed_at":"2026-08-09T18:15:32.929824Z","title":"Removing rlhf protections in gpt-4 via fine-tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.929824Z"},"links":{"cited_paper":"/paper/2311.05553","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:289223fe6651604146eda01fc280cade1510abb33bc143ae3cd00984ba2deee8","observation_id":"c455825a-9c42-49be-b81d-cdd36d81e952","resolution":{"observed_at":"2026-08-09T18:15:32.929824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-09T18:15:33.174855Z","title":"Towards comprehensive and efficient post safety alignment of large language models via safety patching, 2024","venue":null,"work_id":"76521318-d66a-4c7e-a0f4-374fa61905dd","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.933102Z"},"links":{"citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:2b188c1a73831a99321a1ca46321809414d525cd2478b6def5de94eef204d28a","observation_id":"8951279a-4609-4e0c-a8b0-207e8a98574d","resolution":{"observed_at":"2026-08-09T18:15:33.178303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12343","last_updated":"2024-06-06T12:54:48Z","snapshot_observed_at":"2026-08-10T22:54:13.233449Z","submitted_at":"2024-02-19T18:16:51Z","title":"Emulated Disalignment: Safety Alignment for Large Language Models May Backfire!","version":4},"cited_work":{"arxiv_id":"2402.12343","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.12343","snapshot_observed_at":"2026-08-09T18:15:32.966420Z","title":"Emulated Disalignment: Safety Alignment for Large Language Models May Backfire!","venue":"cs.CL","work_id":"ec58ef0b-81c3-42b6-a2d2-a0efe20a7d46","year":2024},"citing_paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-09T18:15:32.936227Z"},"links":{"cited_paper":"/paper/2402.12343","citing_paper":"/paper/2502.00669"},"observation_digest":"sha256:fd253eefc602a1e1cb5b231d2dfd83a1a63a8a65e872b344f0e6a0716b66153c","observation_id":"97daa6df-ba0d-4f32-81f6-2bc62ef2573d","resolution":{"observed_at":"2026-08-09T18:15:32.972047Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.00669","last_updated":"2025-02-02T04:43:35Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T11:04:32.881389Z","submitted_at":"2025-02-02T04:43:35Z","title":"Safety Alignment Depth in Large Language Models: A Markov Chain Perspective"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":26,"verified_exact":2,"verified_fuzzy":10},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 0 inbound Pith citation observations for arXiv:2502.00669."}