{"as_of":"2026-08-14T19:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:486e5e8c7ab224aeeae04607d40b21a3651577b48859be29cc6b18815872b746","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":24,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T13:29:00.171371Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:29:45.394907Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T12:35:25.837701Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24473","last_updated":"2025-06-05T11:42:24Z","snapshot_observed_at":"2026-08-14T12:36:26.988141Z","submitted_at":"2025-05-30T11:20:44Z","title":"Train One Sparse Autoencoder Across Multiple Sparsity Budgets to Preserve Interpretability and Accuracy","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:25.837701Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2505.24473"},"observation_digest":"sha256:5774d8e7a3f4b559f14e1a4418465d1a4c8d3a8da5bf66b9232fedd2e210635a","observation_id":"82a4d355-2fc2-4888-a10d-f5e945759d3e","resolution":{"observed_at":"2026-08-07T12:35:25.837701Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T11:56:15.179663Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01197","last_updated":"2025-06-01T22:20:07Z","snapshot_observed_at":"2026-08-10T06:48:33.503354Z","submitted_at":"2025-06-01T22:20:07Z","title":"Incorporating Hierarchical Semantics in Sparse Autoencoder Architectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:15.179663Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2506.01197"},"observation_digest":"sha256:a2ec4113fc6da8ef3586f77601d8ad3218664ac8e247521a111d14188c300236","observation_id":"19903c6d-936d-43b0-9afe-7f2362cf3100","resolution":{"observed_at":"2026-08-07T11:56:15.179663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-07T04:45:42.719642Z","title":"Woosuk Kwon, Zhuohan Li, Siyuan Zhuang, Ying Sheng, Lianmin Zheng, Cody Hao Yu, Joseph E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09967","last_updated":"2025-06-13T18:01:49Z","snapshot_observed_at":"2026-08-13T21:32:14.045906Z","submitted_at":"2025-06-11T17:44:01Z","title":"Resa: Transparent Reasoning Models via SAEs","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:45:42.719642Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2506.09967"},"observation_digest":"sha256:089bb847a55a6c4d1c141e5e6e05fee0f60ff48c1ce0df395b068a47eeae975d","observation_id":"620053fa-a4fe-44e0-a804-5c35444fc6ec","resolution":{"observed_at":"2026-08-07T04:45:42.719642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-06T18:24:51.180244Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08473","last_updated":"2025-07-11T10:31:53Z","snapshot_observed_at":"2026-08-14T09:25:51.088740Z","submitted_at":"2025-07-11T10:31:53Z","title":"Evaluating SAE interpretability without explanations","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:24:51.180244Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2507.08473"},"observation_digest":"sha256:00b16224c4abe94f5cac9b874c5b4751708d7a7d0ff49131dd17f1101a8655b8","observation_id":"0a50bec3-c959-4b47-af7f-34a36a929314","resolution":{"observed_at":"2026-08-06T18:24:51.180244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-06T15:24:45.744009Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability, 2025 a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.15977","last_updated":"2025-07-21T18:17:18Z","snapshot_observed_at":"2026-08-08T15:45:05.019347Z","submitted_at":"2025-07-21T18:17:18Z","title":"On the transferability of Sparse Autoencoders for interpreting compressed models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T15:24:45.744009Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2507.15977"},"observation_digest":"sha256:36d5ae94f37c07cd2877e0d4c68889541396536136c924399213db1d726c8735","observation_id":"4112d7ac-bf41-414e-88b3-355f5663fc65","resolution":{"observed_at":"2026-08-06T15:24:45.744009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-05T14:27:25.439860Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.21324","last_updated":"2025-08-29T04:42:17Z","snapshot_observed_at":"2026-08-09T09:00:42.140024Z","submitted_at":"2025-08-29T04:42:17Z","title":"Distribution-Aware Feature Selection for SAEs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T14:27:25.439860Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2508.21324"},"observation_digest":"sha256:a83c58fb3ebef9f4186de5fdfed5a277b2d9fda15596d36aac931fde347e6626","observation_id":"5c41650a-07f1-43c6-bec8-c9c5004543f8","resolution":{"observed_at":"2026-08-05T14:27:25.439860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2509.18127","last_updated":"2026-04-14T10:00:44Z","snapshot_observed_at":"2026-08-02T17:25:00.233548Z","submitted_at":"2025-09-11T11:22:43Z","title":"Safe-SAIL: Towards a Fine-grained Safety Landscape of Large Language Models via Sparse Autoencoder Interpretation Framework","version":3},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-18T18:13:01.662828Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2509.18127"},"observation_digest":"sha256:3ff1bffc799cebfb270abde72f643e32837f2d317651dafbc6b90872d7e5816d","observation_id":"00bc17a8-ec58-433b-9b25-7291e4291f8c","resolution":{"observed_at":"2026-05-18T18:16:43.748207Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2601.14004","last_updated":"2026-04-14T03:49:06Z","snapshot_observed_at":"2026-08-11T17:12:54.016914Z","submitted_at":"2026-01-20T14:23:23Z","title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","version":4},"reference_index":148,"source":"pdf_text","source_observed_at":"2026-05-16T12:39:57.398423Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2601.14004"},"observation_digest":"sha256:e6baad982d37c842a2ce2264447cbd5e593f6c0d70e24c80e6e1a9f2dad539a2","observation_id":"eb0d2f34-a1c1-4baa-98c7-71091dd817b0","resolution":{"observed_at":"2026-05-16T12:40:54.605878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-02T18:56:20.000242Z","title":"Leo Gao, Tom Dupré la Tour, Henk Tillman, Gabriel Goh, Rajan Troll, Alec Radford, Ilya Sutskever, Jan Leike, and Jeffrey Wu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.04198","last_updated":"2026-06-16T16:00:25Z","snapshot_observed_at":"2026-08-13T05:33:13.764056Z","submitted_at":"2026-03-04T15:46:23Z","title":"Stable and Steerable Sparse Autoencoders with Weight Regularization","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T18:56:20.000242Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2603.04198"},"observation_digest":"sha256:564f07b204537bd2e8292df2984aa839c8e02abdc0fcb2bb652f823c45073f41","observation_id":"1e7376d5-7454-4840-8813-48c7a2539620","resolution":{"observed_at":"2026-08-02T18:56:20.000242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2604.08846","last_updated":"2026-04-10T01:01:56Z","snapshot_observed_at":"2026-08-13T19:19:16.724751Z","submitted_at":"2026-04-10T01:01:56Z","title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T18:04:05.157103Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2604.08846"},"observation_digest":"sha256:fca4c0aff414c364946679a27882e06f160e2dcfebb73d055432c88ffea50740","observation_id":"f372237d-5f3a-434a-9a9c-253ec13b55e8","resolution":{"observed_at":"2026-05-11T05:35:57.627704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.05223","last_updated":"2026-04-18T05:53:31Z","snapshot_observed_at":"2026-08-12T20:36:42.610000Z","submitted_at":"2026-04-18T05:53:31Z","title":"Structural Instability of Feature Composition","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T06:40:00.507484Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.05223"},"observation_digest":"sha256:2e6c72023d11a6e0ac879c89b185a13b9a8bc2a72f65ad184ab597d444747fb7","observation_id":"41ce46e6-8ce3-437d-9d3d-838688b499cf","resolution":{"observed_at":"2026-05-10T06:41:36.519790Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.06494","last_updated":"2026-05-07T16:15:16Z","snapshot_observed_at":"2026-08-13T15:40:07.078663Z","submitted_at":"2026-05-07T16:15:16Z","title":"From Token Lists to Graph Motifs: Weisfeiler-Lehman Analysis of Sparse Autoencoder Features","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T09:41:01.775116Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.06494"},"observation_digest":"sha256:3f3db8b5696ffa200e3e6e92cf9520883a8b3ce4cd0e0eae2bab7f690f5ae5ec","observation_id":"fc8f724e-4403-4fbb-96fc-6f84e4825e25","resolution":{"observed_at":"2026-05-11T20:16:11.060148Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.07922","last_updated":"2026-05-11T02:18:14Z","snapshot_observed_at":"2026-08-13T10:23:38.689080Z","submitted_at":"2026-05-08T15:57:37Z","title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T03:13:58.543525Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.07922"},"observation_digest":"sha256:c7d91a245d1502d0081dd7a9bb0f64d5cfc150471d8fac317a00a7bce5b2197a","observation_id":"12da451f-70cb-4dca-a9f9-1ce293485063","resolution":{"observed_at":"2026-05-11T03:15:54.303210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.07922","last_updated":"2026-05-11T02:18:14Z","snapshot_observed_at":"2026-08-13T10:23:38.689080Z","submitted_at":"2026-05-08T15:57:37Z","title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-12T03:35:50.776347Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.07922"},"observation_digest":"sha256:8e8e74f6c9eff6933086db0786f510d93ebd47963fec4d647df1a8adda502c1d","observation_id":"793d19d5-c2a0-4e38-a5d5-813223894e8a","resolution":{"observed_at":"2026-05-12T07:16:25.445401Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.10536","last_updated":"2026-05-11T13:19:45Z","snapshot_observed_at":"2026-08-08T17:01:17.876100Z","submitted_at":"2026-05-11T13:19:45Z","title":"HH-SAE: Discovering and Steering Hierarchical Knowledge of Complex Manifolds","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T03:13:53.559096Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.10536"},"observation_digest":"sha256:6de219bd2135b3650e40f7a253c67b1a0f0707ad0dc4ce75b2c0d0ce84497e49","observation_id":"47bdda45-8ee3-4658-8620-e9586d64adf1","resolution":{"observed_at":"2026-05-12T03:16:19.139898Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T23:09:28.260788Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T14:16:44.232080Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:55d1e7a385cc68b683d4da81254d86e5a4cbcd80bd0eeac34ffb618489b5e74e","observation_id":"6ec363f6-219e-438e-a626-a06e5bd5fef3","resolution":{"observed_at":"2026-06-29T14:23:30.730641Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-04T05:02:51.380352Z","title":"SAEBench: A comprehensive benchmark for sparse autoencoders in language model interpretability.arXiv preprint arXiv:2503.09532, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.28149","last_updated":"2026-08-04T06:19:42Z","snapshot_observed_at":"2026-08-07T23:09:28.260788Z","submitted_at":"2026-05-27T08:31:43Z","title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T05:02:51.380352Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2605.28149"},"observation_digest":"sha256:cbb0dd6484a147374c04044dd93dcc08f8a4c4e7f79b712c0673678ea0eef962","observation_id":"a84392a0-db3d-4784-95e6-381b677202d3","resolution":{"observed_at":"2026-08-04T05:02:51.380352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2606.09653","last_updated":"2026-06-08T15:42:39Z","snapshot_observed_at":"2026-08-12T16:34:06.376333Z","submitted_at":"2026-06-08T15:42:39Z","title":"A Unifying Framework for Concept-Based Representational Similarity","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T17:10:36.855674Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2606.09653"},"observation_digest":"sha256:4e62fcd5e6fe74d4221f9b11b48f211d54fd8ccbed4460de83e6eca8f68d14ed","observation_id":"4091bcb3-81ce-41b3-8e70-c13ca7a8f2df","resolution":{"observed_at":"2026-07-03T00:27:29.987745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":"2503.09532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-07-04T10:29:45.394907Z","title":"Saebench: A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":"62a3f4ed-bd4a-4b51-b46c-52be11356293","year":2025},"citing_paper":{"arxiv_id":"2606.22994","last_updated":"2026-06-22T08:12:34Z","snapshot_observed_at":"2026-08-08T15:52:49.568637Z","submitted_at":"2026-06-22T08:12:34Z","title":"Do Sparse Autoencoders Learn Meaningful Concept Hierarchies?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-26T08:46:48.220801Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2606.22994"},"observation_digest":"sha256:68b05fa1e1abcc5f7815f20aae54219565bd06d17c5b69cc7bdf3b0b0ec447bd","observation_id":"2e6d96a4-bc02-44ab-b873-98d8a489405f","resolution":{"observed_at":"2026-07-04T10:29:45.396552Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T19:02:29.759226Z","title":"doi:10.48550/arXiv.2503.09532 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17117","last_updated":"2026-07-19T08:07:32Z","snapshot_observed_at":"2026-08-09T17:59:45.832097Z","submitted_at":"2026-07-19T08:07:32Z","title":"Persistent Sparse Autoencoders: Learning Feature Timescales in Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-01T19:02:29.759226Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.17117"},"observation_digest":"sha256:6a67b838b31e33ce8ae320f00ca22dd3c0c3e5844e0bd6f0e665a05f4f34aa9c","observation_id":"054b1070-4bea-4917-a1ad-4f0284afe438","resolution":{"observed_at":"2026-08-01T19:02:29.759226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T18:04:18.007688Z","title":"2503.09532 , archiveprefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17425","last_updated":"2026-07-21T01:20:50Z","snapshot_observed_at":"2026-08-09T20:58:05.060704Z","submitted_at":"2026-07-19T22:19:17Z","title":"Decoder-Preserving Sparse Autoencoders: Which Readouts Survive Sparse Compression?","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-01T18:04:18.007688Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.17425"},"observation_digest":"sha256:31822476b4801ca491626d2f4c7a8feb46229b47b8881c24392a7aeb8a34e7f2","observation_id":"c7c1800e-f602-48f6-ab03-8968566103d0","resolution":{"observed_at":"2026-08-01T18:04:18.007688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T10:03:56.301338Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20596","last_updated":"2026-07-22T17:33:16Z","snapshot_observed_at":"2026-08-09T06:27:26.743612Z","submitted_at":"2026-07-22T17:33:16Z","title":"Are Single-Token Sparse Autoencoder Features Causally Necessary? Layer-Depth and SAE-Family Effects","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-01T10:03:56.301338Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.20596"},"observation_digest":"sha256:138f0d5c46761967b241ae8530a911d5b7853630c2a9a497dda01fb6223ed5e6","observation_id":"b87c7a6c-e658-41fd-ac48-2b0833781747","resolution":{"observed_at":"2026-08-01T10:03:56.301338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-01T07:58:32.246258Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.27404","last_updated":"2026-07-29T19:20:50Z","snapshot_observed_at":"2026-08-07T16:37:14.209504Z","submitted_at":"2026-07-29T19:20:50Z","title":"ECG-InterpBench: Benchmarking the Interpretability of ECG Foundation Models with Matched-Scale Sparse Autoencoders","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T07:58:32.246258Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2607.27404"},"observation_digest":"sha256:368c73c132a7e529692c15542c28ce4def3e12a8debf887319ba41493ee20d40","observation_id":"959c6a7f-12e1-46eb-8ee7-0a5a4c151dd9","resolution":{"observed_at":"2026-08-01T07:58:32.246258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09532","snapshot_observed_at":"2026-08-14T13:29:00.171371Z","title":"SAEBench : A comprehensive benchmark for sparse autoencoders in language model interpretability","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13337","last_updated":"2026-08-13T15:06:56Z","snapshot_observed_at":"2026-08-14T18:31:47.290809Z","submitted_at":"2026-08-13T15:06:56Z","title":"Where You Measure Decides What You Measure: Position Selection in Ablation-Based SAE Evaluation","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-14T13:29:00.171371Z"},"links":{"cited_paper":"/paper/2503.09532","citing_paper":"/paper/2608.13337"},"observation_digest":"sha256:1d77310bd1284c858f33ee19219046eae19a424ce4970f4953fd7c32f6927214","observation_id":"cbbd8f16-e1d9-4f43-a074-7d7f632f70ea","resolution":{"observed_at":"2026-08-14T13:29:00.171371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.09532/citation-record","integrity":"/paper/2503.09532/integrity","json":"/paper/2503.09532/citation-record.json","paper":"/paper/2503.09532"},"outbound":[],"paper":{"arxiv_id":"2503.09532","last_updated":"2025-06-04T13:27:28Z","latest_version":4,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T03:23:02.234579Z","submitted_at":"2025-03-12T16:49:02Z","title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 24 inbound Pith citation observations for arXiv:2503.09532."}