{"as_of":"2026-08-20T09:31:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7de4d7fcb4ce9295ae44bbe1794524ae77ebc543f34d32d15f19ae951fe96270","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T12:19:44.215403Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":10,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2304.05969","last_updated":"2023-05-16T16:24:55Z","snapshot_observed_at":"2026-08-15T09:56:57.369509Z","submitted_at":"2023-04-12T16:46:43Z","title":"Localizing Model Behavior with Path Patching","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-16T19:38:37.751487Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2304.05969"},"observation_digest":"sha256:779b13bfb74095cf41a5ef8b305dc92aa30aa46f85fd2999d3fe11bc059eb866","observation_id":"f2e48289-ffce-497a-9a7b-aa5db8fe4d98","resolution":{"observed_at":"2026-05-16T19:38:37.803414Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2310.15154","last_updated":"2023-10-23T17:55:31Z","snapshot_observed_at":"2026-08-08T01:44:57.952179Z","submitted_at":"2023-10-23T17:55:31Z","title":"Linear Representations of Sentiment in Large Language Models","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-15T12:43:06.361170Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2310.15154"},"observation_digest":"sha256:efb0baa70efa823b64962fd2bccee821420eaccc40a3d3df1e478c63826314f7","observation_id":"ac85d2ad-2034-4509-b7ff-a108d493aa27","resolution":{"observed_at":"2026-05-15T12:43:06.462170Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2403.19647","last_updated":"2025-03-27T05:44:45Z","snapshot_observed_at":"2026-08-14T07:35:01.333549Z","submitted_at":"2024-03-28T17:56:07Z","title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","version":3},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-13T13:15:10.632115Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2403.19647"},"observation_digest":"sha256:c204c38be50fa29aed83b79375cb63376b92136557d35c9f46b9269de8bbe106","observation_id":"072600d1-7c4a-4cc7-98f8-874fb6884ae6","resolution":{"observed_at":"2026-05-13T13:15:10.703650Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-11T23:28:45.301776Z","title":"Causal abstraction for faithful model interpretation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02579","last_updated":"2024-12-20T18:38:53Z","snapshot_observed_at":"2026-08-14T21:34:03.233913Z","submitted_at":"2024-12-03T17:04:20Z","title":"Factored space models: Towards causality between levels of abstraction","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T23:28:45.301776Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2412.02579"},"observation_digest":"sha256:83343b4916a8c28e8147116508cd3deee8a1d955b4a63a89dd7c6b7f03368311","observation_id":"e44f166a-e856-466f-9106-7dacbddeacb6","resolution":{"observed_at":"2026-08-11T23:28:45.301776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-11T23:03:10.792600Z","title":"@7N 37^3b#+]xI[bb ]CxAFgGxx=_ l ЛBPK 2Fi tY # W w(( qx 'ҋ9y |,BP","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02893","last_updated":"2024-12-03T22:58:21Z","snapshot_observed_at":"2026-08-14T07:35:41.182066Z","submitted_at":"2024-12-03T22:58:21Z","title":"Removing Spurious Correlation from Neural Network Interpretations","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T23:03:10.792600Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2412.02893"},"observation_digest":"sha256:ed4e96fb1b543e8449ce7e6048508a543fd27bd8c28a7db4ca387cd823cd43c8","observation_id":"07fe9ccb-2b20-411f-ba91-3906b2393cb6","resolution":{"observed_at":"2026-08-11T23:03:10.792600Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-09T20:38:36.470902Z","title":"Green and grue causal variables","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2501.19335","last_updated":"2025-02-03T17:24:50Z","snapshot_observed_at":"2026-08-15T23:28:38.238995Z","submitted_at":"2025-01-31T17:35:21Z","title":"What is causal about causal models and representations?","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-09T20:38:36.470902Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2501.19335"},"observation_digest":"sha256:1713b64427a94b268366571352be3d559fa6bfe42bb4cbe5da730458fe4b27b7","observation_id":"a94c6dd8-759c-4b98-b19c-1e391c6f6b8e","resolution":{"observed_at":"2026-08-09T20:38:36.470902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-16T12:19:44.215403Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.13151","last_updated":"2025-06-09T17:44:50Z","snapshot_observed_at":"2026-08-17T22:04:07.354628Z","submitted_at":"2025-04-17T17:55:45Z","title":"MIB: A Mechanistic Interpretability Benchmark","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T12:19:44.215403Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2504.13151"},"observation_digest":"sha256:03915471206e6f433039e1820d35b71a9c2916fa9193e3e00a000641ae1b20c0","observation_id":"0d7cd494-b1de-4b9b-b4bd-199414ac0a7c","resolution":{"observed_at":"2026-08-16T12:19:44.215403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-16T04:25:05.733991Z","title":"Causal abstraction for faithful model interpretation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-19T03:13:58.617120Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.733991Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:bb257df63e06b21a497365b2dabbaa912ee1e125cd9c9b3049d02d080c70108c","observation_id":"22aea982-0987-44f7-ab5b-592ababe55dd","resolution":{"observed_at":"2026-08-16T04:25:05.733991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-15T20:41:57.678412Z","title":"Transformer Circuits Thread","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.12268","last_updated":"2025-06-05T01:45:59Z","snapshot_observed_at":"2026-08-17T19:25:30.922752Z","submitted_at":"2025-05-18T07:15:01Z","title":"$K$-MSHC: Unmasking Minimally Sufficient Head Circuits in Large Language Models with Experiments on Syntactic Classification Tasks","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T20:41:57.678412Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.12268"},"observation_digest":"sha256:a8842bd5cc244f8bd0d3fe405c2ae54a7d272aa7936842d6305ded9e3e83d0ac","observation_id":"831539e5-aef9-43d1-9f5a-1fa53e94a015","resolution":{"observed_at":"2026-08-15T20:41:57.678412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-07T15:43:31.583205Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14300","last_updated":"2026-07-10T00:11:31Z","snapshot_observed_at":"2026-08-17T11:12:54.898690Z","submitted_at":"2025-05-20T12:49:58Z","title":"Beyond Black-Box Obfuscation: Mechanistic Analysis and Defense of White-Box Monitors","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T15:43:31.583205Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.14300"},"observation_digest":"sha256:76190c294922d8896b191ab333602e28fb8ea8438cfd1ae241676855199d58db","observation_id":"60e2b358-386b-40d1-b04e-197bf857f57f","resolution":{"observed_at":"2026-08-07T15:43:31.583205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-07T15:42:40.162144Z","title":"Geiger, D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14424","last_updated":"2025-05-20T14:32:03Z","snapshot_observed_at":"2026-08-13T09:33:21.471790Z","submitted_at":"2025-05-20T14:32:03Z","title":"Explaining Neural Networks with Reasons","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:40.162144Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.14424"},"observation_digest":"sha256:7ce151e9ce387df45c1fca10a97fec47c74da20d5aeabd0dbde6eeec0ac61133","observation_id":"8240af5a-2aed-4c27-985e-c4e89b35704f","resolution":{"observed_at":"2026-08-07T15:42:40.162144Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-07T14:03:00.147887Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability.arXiv preprint arXiv:2301.04709, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.20254","last_updated":"2025-05-26T17:31:36Z","snapshot_observed_at":"2026-08-15T08:10:04.533805Z","submitted_at":"2025-05-26T17:31:36Z","title":"Position: Mechanistic Interpretability Should Prioritize Feature Consistency in SAEs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:03:00.147887Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.20254"},"observation_digest":"sha256:8029dd1e4546701951570a2527f1159eb8fd19290cb686b0d672cc066066dc12","observation_id":"ed653d77-400e-476f-b68f-b3871cd5a005","resolution":{"observed_at":"2026-08-07T14:03:00.147887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-07T13:50:07.811798Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.20896","last_updated":"2025-05-30T18:08:50Z","snapshot_observed_at":"2026-08-20T08:34:41.133399Z","submitted_at":"2025-05-27T08:39:20Z","title":"How Do Transformers Learn Variable Binding in Symbolic Programs?","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T13:50:07.811798Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.20896"},"observation_digest":"sha256:82fc61e59fad9496b6271ff2d2f95f431995263a63c8a643dbf4b678d339470b","observation_id":"60b65955-83ca-44f0-98ca-bf7a822f87cd","resolution":{"observed_at":"2026-08-07T13:50:07.811798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-06T22:16:46.623506Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.22105","last_updated":"2025-06-27T10:35:41Z","snapshot_observed_at":"2026-08-09T02:44:01.134272Z","submitted_at":"2025-06-27T10:35:41Z","title":"Identifying a Circuit for Verb Conjugation in GPT-2","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T22:16:46.623506Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2506.22105"},"observation_digest":"sha256:3419735cba0db323323ca27c0da46275f0f574946ae5f57d85182f4e83b654b0","observation_id":"a99928f2-fe36-4677-b909-72e3eae049fc","resolution":{"observed_at":"2026-08-06T22:16:46.623506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-06T19:22:28.725891Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06213","last_updated":"2025-07-08T17:46:08Z","snapshot_observed_at":"2026-08-09T00:29:57.535949Z","submitted_at":"2025-07-08T17:46:08Z","title":"Identifiability in Causal Abstractions: A Hierarchy of Criteria","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T19:22:28.725891Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2507.06213"},"observation_digest":"sha256:be0a5adeb3f3c38eab4b094d54a9a2fc6a17a6a99d85c9c145ff767726c46129","observation_id":"88fa77e4-1605-455c-a09d-0b79149d539c","resolution":{"observed_at":"2026-08-06T19:22:28.725891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-06T19:10:28.638962Z","title":"Geiger, D","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06445","last_updated":"2026-07-21T16:59:32Z","snapshot_observed_at":"2026-08-18T18:34:24.314623Z","submitted_at":"2025-07-08T23:07:33Z","title":"Can Interpretation Predict Behavior on Unseen Data?","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T19:10:28.638962Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2507.06445"},"observation_digest":"sha256:f301e07321b0874432c64ba7272556bb856df6fb1def9ac757c9b66f1616a1d1","observation_id":"0e43d9ef-db7d-46cd-aa37-b05d7851210e","resolution":{"observed_at":"2026-08-06T19:10:28.638962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-06T15:51:37.348803Z","title":"Causal abstraction for faithful model interpretation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14901","last_updated":"2025-07-20T10:25:24Z","snapshot_observed_at":"2026-08-17T13:00:39.414924Z","submitted_at":"2025-07-20T10:25:24Z","title":"Learning Nonlinear Causal Reductions to Explain Reinforcement Learning Policies","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T15:51:37.348803Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2507.14901"},"observation_digest":"sha256:ef571df99f6b21b10e30d195e50dcfee1cabd595335705879f4c190343a5d4a1","observation_id":"4ecdeea4-ca1e-4230-8825-1855bc0ca1db","resolution":{"observed_at":"2026-08-06T15:51:37.348803Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-15T18:21:12.926237Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22928","last_updated":"2025-07-24T10:25:46Z","snapshot_observed_at":"2026-08-19T21:05:52.525711Z","submitted_at":"2025-07-24T10:25:46Z","title":"How does Chain of Thought Think? Mechanistic Interpretability of Chain-of-Thought Reasoning with Sparse Autoencoding","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-15T18:21:12.926237Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2507.22928"},"observation_digest":"sha256:1a018f5451a9d7f13305efa79ede9526cbe2cf5adcfe2d9c6ba966aa1155dd69","observation_id":"d7dee443-348d-4baa-98ad-9c9fe9f2a572","resolution":{"observed_at":"2026-08-15T18:21:12.926237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.01164","last_updated":"2026-05-01T23:46:29Z","snapshot_observed_at":"2026-08-12T12:29:50.349827Z","submitted_at":"2026-05-01T23:46:29Z","title":"LLMs Should Not Yet Be Credited with Decision Explanation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-09T18:41:24.552939Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.01164"},"observation_digest":"sha256:5232589cd1580699736ff55f0f28765ebe756ba21908807b174dd53d407a8fa6","observation_id":"03fbbe87-c027-441e-a4b8-17876f48c024","resolution":{"observed_at":"2026-05-11T16:06:33.196776Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.02234","last_updated":"2026-05-04T05:09:21Z","snapshot_observed_at":"2026-08-15T08:37:33.356655Z","submitted_at":"2026-05-04T05:09:21Z","title":"Bucketing the Good Apples: A Method for Diagnosing and Improving Causal Abstraction","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-08T19:12:30.627633Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.02234"},"observation_digest":"sha256:74196b049eed7b0eab5df5a010cb44e7edf792e9f600f4b8dec3755bd7cef203","observation_id":"38219848-7677-43dd-aed7-9a0cbec2407f","resolution":{"observed_at":"2026-05-09T06:00:36.290694Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.08934","last_updated":"2026-06-16T20:17:16Z","snapshot_observed_at":"2026-08-11T08:05:03.537052Z","submitted_at":"2026-05-09T13:08:07Z","title":"From Mechanistic to Compositional Interpretability","version":1},"reference_index":147,"source":"arxiv_source","source_observed_at":"2026-05-12T02:42:26.173782Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.08934"},"observation_digest":"sha256:e2ac23a4c304be33b7b5e5a6f6d132e020b889249a81addb2cc15ded49ceb28d","observation_id":"9395ec40-e26a-4326-91d3-68fa7c649b7c","resolution":{"observed_at":"2026-05-12T02:46:18.264696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.12809","last_updated":"2026-05-12T23:01:29Z","snapshot_observed_at":"2026-08-15T05:36:43.446023Z","submitted_at":"2026-05-12T23:01:29Z","title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-14T20:17:01.224864Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.12809"},"observation_digest":"sha256:36ccc082faa4ab92acd643040ebc8d3a3ecd762217ed46c272b188429e4fc505","observation_id":"dcc8bb27-776b-4289-895d-4364afb81e32","resolution":{"observed_at":"2026-05-14T20:17:54.151029Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.15328","last_updated":"2026-05-14T18:41:20Z","snapshot_observed_at":"2026-08-15T03:26:56.258016Z","submitted_at":"2026-05-14T18:41:20Z","title":"From Weight Perturbation to Feature Attribution for Explaining Fully Connected Neural Networks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T16:06:05.508610Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.15328"},"observation_digest":"sha256:bb5fc03f39f65107bf088acca9aedc4310e880eb8d585b2244e4424c68ae1e7a","observation_id":"540a2dd8-b6c1-4a46-b7d8-b72425f6549b","resolution":{"observed_at":"2026-05-19T16:07:41.081855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.25891","last_updated":"2026-05-25T14:19:51Z","snapshot_observed_at":"2026-08-15T14:52:18.833155Z","submitted_at":"2026-05-25T14:19:51Z","title":"Causal Tongue-Tie: LLMs Can Encode Causal Direction, But Their Yes/No Outputs Fail to Express","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-29T22:13:32.378862Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.25891"},"observation_digest":"sha256:3a435c6ad7992915804e16affe9e7c6ab3b472f32b128f3a0e67954221927986","observation_id":"e0173868-2849-46b3-9fed-ba9a2a8aec7b","resolution":{"observed_at":"2026-06-29T22:13:59.318138Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2605.26431","last_updated":"2026-07-17T14:23:48Z","snapshot_observed_at":"2026-08-13T08:40:35.532894Z","submitted_at":"2026-05-26T01:36:41Z","title":"Probing LLMs for Syntactic Structure Beyond Universal Dependencies: A Minimalist Phase Account in English","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-29T18:44:23.296625Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.26431"},"observation_digest":"sha256:0265a8d81b3b00b59f172292503b1221d9023a17c206033aaa78f5f45d67d7d2","observation_id":"82492bf6-f21a-4999-941f-563b442d697d","resolution":{"observed_at":"2026-06-29T18:53:51.895670Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-02T13:14:20.387389Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.26431","last_updated":"2026-07-17T14:23:48Z","snapshot_observed_at":"2026-08-13T08:40:35.532894Z","submitted_at":"2026-05-26T01:36:41Z","title":"Probing LLMs for Syntactic Structure Beyond Universal Dependencies: A Minimalist Phase Account in English","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-02T13:14:20.387389Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2605.26431"},"observation_digest":"sha256:82a25b15f1e7065371b2a0e8e60659f32d9a446d8b0080a6a07871c2c4891e36","observation_id":"9dbb2565-b89d-4eed-9a79-7e1331e76973","resolution":{"observed_at":"2026-08-02T13:14:20.387389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.05194","last_updated":"2026-07-08T19:54:16Z","snapshot_observed_at":"2026-08-06T06:51:53.907927Z","submitted_at":"2026-05-11T21:09:00Z","title":"Temporal Preference Concepts and their Functions in a Large Language Model","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-30T22:16:47.743387Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.05194"},"observation_digest":"sha256:634b27fdfe8383dbc4f40a282f7b4f43aca57f84ed9d788d668129afb6fa9517","observation_id":"c934caf3-ffe0-40ac-961f-61d6a5a5394a","resolution":{"observed_at":"2026-07-01T14:05:47.137704Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-07-12T17:03:44.315006Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.05194","last_updated":"2026-07-08T19:54:16Z","snapshot_observed_at":"2026-08-06T06:51:53.907927Z","submitted_at":"2026-05-11T21:09:00Z","title":"Temporal Preference Concepts and their Functions in a Large Language Model","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T17:03:44.315006Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.05194"},"observation_digest":"sha256:87c3c32b37930ce11862fd6d8011d75acf87dac1a8c275548a147deb3b243b33","observation_id":"7b896aa9-f0cd-4ff4-8612-394cf17fddc8","resolution":{"observed_at":"2026-07-12T17:03:44.315006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.06840","last_updated":"2026-06-05T02:32:24Z","snapshot_observed_at":"2026-08-16T20:02:22.679518Z","submitted_at":"2026-06-05T02:32:24Z","title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-06-27T22:22:52.690010Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.06840"},"observation_digest":"sha256:7c67e1ec5d2f875965a41c78d3c56aeed0455005e3272933859fea631041c554","observation_id":"f2455939-37e6-42cb-8548-afb76926aa3b","resolution":{"observed_at":"2026-06-27T22:31:21.287226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.08044","last_updated":"2026-08-04T12:50:00Z","snapshot_observed_at":"2026-08-07T23:11:38.433528Z","submitted_at":"2026-06-06T08:10:56Z","title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-27T20:04:17.744876Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.08044"},"observation_digest":"sha256:86d6aacc76318d4014465d94800bb96eee8639214c26b4a513c1f52153bb411d","observation_id":"6204ff38-c9a2-4847-8cf6-6787df2f3870","resolution":{"observed_at":"2026-07-02T20:57:23.074500Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.10877","last_updated":"2026-06-09T13:52:05Z","snapshot_observed_at":"2026-08-16T13:34:24.704519Z","submitted_at":"2026-06-09T13:52:05Z","title":"XtrAIn: Training-Guided Occlusion for Feature Attribution","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T13:37:35.691503Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.10877"},"observation_digest":"sha256:1c33bb40c75112934cc7b95948e822ebf81aee09aa0404bc63ffe5162a58b0aa","observation_id":"201b7313-9d69-4b6e-b126-58c5af926390","resolution":{"observed_at":"2026-07-03T04:47:38.382726Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.19741","last_updated":"2026-06-18T03:14:10Z","snapshot_observed_at":"2026-08-08T12:44:09.477104Z","submitted_at":"2026-06-18T03:14:10Z","title":"Interpreting Neural Combinatorial Optimization via Evolving Programmatic Bottlenecks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-26T17:49:42.069456Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.19741"},"observation_digest":"sha256:b5a43b146090b8757bccd0b4aa5bf370777c1f1d4dbe525987c0b3f8ccbe1a6c","observation_id":"832bb100-a667-4f0d-adbb-bf5ed7e9d6c0","resolution":{"observed_at":"2026-07-04T03:39:30.217596Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.25657","last_updated":"2026-06-24T10:11:04Z","snapshot_observed_at":"2026-08-10T18:37:42.196432Z","submitted_at":"2026-06-24T10:11:04Z","title":"Steering Vision-Language Models with Joint Sparse Autoencoders","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-06-25T20:56:57.716246Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.25657"},"observation_digest":"sha256:7f994ea97072ed8b5133fada17271b3f820d660d3d80c667e0bff218bf9ce282","observation_id":"c6fe8413-40de-4ea3-aecc-856b9dbde6f4","resolution":{"observed_at":"2026-07-04T20:00:07.779400Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":"2301.04709","doi":"10.48550/arxiv.2301.04709","metadata_source":"arxiv_reference","pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Causal abstraction for faithful model interpretation","venue":"arXiv (Cornell University)","work_id":"22433217-97ea-4106-96b3-fbded5b74482","year":2023},"citing_paper":{"arxiv_id":"2606.29657","last_updated":"2026-07-10T21:53:01Z","snapshot_observed_at":"2026-07-16T23:17:54.967525Z","submitted_at":"2026-06-28T23:55:29Z","title":"Safety from Honesty in a Disinterested AI Predictor","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-30T06:53:19.737408Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.29657"},"observation_digest":"sha256:29720b9ee5e294f433b576b813b315545f0f685518d32092aeaef1d309aad38e","observation_id":"51b25163-ad74-4382-87a1-03216aaa2ef6","resolution":{"observed_at":"2026-06-30T06:54:20.196264Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-07-14T17:02:19.189814Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.29657","last_updated":"2026-07-10T21:53:01Z","snapshot_observed_at":"2026-07-16T23:17:54.967525Z","submitted_at":"2026-06-28T23:55:29Z","title":"Safety from Honesty in a Disinterested AI Predictor","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-07-14T17:02:19.189814Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2606.29657"},"observation_digest":"sha256:7afaba5f4fca11c6ff268bf4c0960f6b14f74b95f72c7163ce16065da0dc66e6","observation_id":"890e0878-d980-4dd5-9b5e-32e7b8e7c48c","resolution":{"observed_at":"2026-07-14T17:02:19.189814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-01T10:03:55.219054Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.20596","last_updated":"2026-07-22T17:33:16Z","snapshot_observed_at":"2026-08-15T09:33:27.353760Z","submitted_at":"2026-07-22T17:33:16Z","title":"Are Single-Token Sparse Autoencoder Features Causally Necessary? Layer-Depth and SAE-Family Effects","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-01T10:03:55.219054Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2607.20596"},"observation_digest":"sha256:daa3ab1434c6b3eef2a1150badff527b517f54bb7845f237aae137a361bab72a","observation_id":"891db3d4-d0ba-45a8-b03d-7375297297fd","resolution":{"observed_at":"2026-08-01T10:03:55.219054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2301.04709/citation-record","integrity":"/paper/2301.04709/integrity","json":"/paper/2301.04709/citation-record.json","paper":"/paper/2301.04709"},"outbound":[],"paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","latest_version":4,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2301.04709."}