{"as_of":"2026-08-14T09:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a63e57b62cd5a857cd1840703dfcfa310a2276ca3f876b9ef3bd96732c5fe2b0","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T21:26:42.281399Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:20:01.640542Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T15:20:08.711872Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"cited_work":{"arxiv_id":"2411.08790","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.08790","snapshot_observed_at":"2026-08-07T15:20:08.711872Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","venue":"cs.LG","work_id":"ce8e1191-3076-4e96-bb92-b540b63df1d1","year":2024},"citing_paper":{"arxiv_id":"2505.15524","last_updated":"2025-05-21T13:50:23Z","snapshot_observed_at":"2026-08-07T23:46:52.705100Z","submitted_at":"2025-05-21T13:50:23Z","title":"Evaluate Bias without Manual Test Sets: A Concept Representation Perspective for LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:20:01.640542Z"},"links":{"cited_paper":"/paper/2411.08790","citing_paper":"/paper/2505.15524"},"observation_digest":"sha256:6b3eeb7e43e3a8507b75f9df7910ded83b9eec07d357f48f4835fa38b1a9a810","observation_id":"7efb8429-131f-4ba5-9ae0-ed4562b10300","resolution":{"observed_at":"2026-08-07T15:20:08.798186Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.08790/citation-record","integrity":"/paper/2411.08790/integrity","json":"/paper/2411.08790/citation-record.json","paper":"/paper/2411.08790"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-08-14T08:45:20.216780Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-12T21:26:42.160078Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.160078Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:7a6ade7d68da4ceb0456bc6bcbc7b4dcd068ccc49c0a35bffc99faaa7cdf2567","observation_id":"abb05878-58e8-4ea9-935f-d3ea7f231ea4","resolution":{"observed_at":"2026-08-12T21:26:42.160078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11717","last_updated":"2024-10-30T18:57:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T16:36:12Z","title":"Refusal in Language Models Is Mediated by a Single Direction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.11717","snapshot_observed_at":"2026-08-12T21:26:42.165950Z","title":"Refusal in language models is mediated by a single direction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.165950Z"},"links":{"cited_paper":"/paper/2406.11717","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:470d76bcf471c33fe0b36054a6df5b61fa9528a72b8ec09c5687302ff9e83b43","observation_id":"f81a1ebb-29a5-44bc-9521-7ac6facaa440","resolution":{"observed_at":"2026-08-12T21:26:42.165950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.171114Z","title":"Towards monosemanticity: Decomposing language models with dictionary learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.171114Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:2b0ac8375712a0161e2a5a511f4b65eecffb1fb630b2da695f2430c6cbdf1d7f","observation_id":"81177b00-329f-4445-923b-a851bd84995b","resolution":{"observed_at":"2026-08-12T21:26:42.171114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.690381Z","title":"Progress update #1 from the GDM mech interp team","venue":null,"work_id":"4c0ee005-2154-42be-99d0-279991acd1b8","year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.176621Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:df7a39c614e9af3ba9bd164972e579a66166e0b96f2dd1f297ca5663d4d77456","observation_id":"71c6461e-4406-48b8-96ba-6adb713ca091","resolution":{"observed_at":"2026-08-12T21:26:42.695410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-08-11T07:21:14.568175Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-12T21:26:42.181551Z","title":"Sparse autoen- coders find highly interpretable features in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.181551Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:c4677b714c9c53566ce6634ef20d558c8978582b72d6fd486423b5e65926b731","observation_id":"e0219068-06dd-4abb-a94b-5075e75f8a87","resolution":{"observed_at":"2026-08-12T21:26:42.181551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00027","last_updated":"2020-12-31T19:00:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-12-31T19:00:10Z","title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2101.00027","snapshot_observed_at":"2026-08-12T21:26:42.186908Z","title":"The Pile: An 800GB dataset of diverse text for language modeling","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.186908Z"},"links":{"cited_paper":"/paper/2101.00027","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:fc02aafd9a3cb075095ee3f4651624d974870fea528cabc70679b31096a95c98","observation_id":"033856ad-c918-4231-8177-75aaf694e5fd","resolution":{"observed_at":"2026-08-12T21:26:42.186908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04093","snapshot_observed_at":"2026-08-12T21:26:42.192559Z","title":"Scaling and evaluating sparse autoencoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.192559Z"},"links":{"cited_paper":"/paper/2406.04093","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:34c813d7f11b8bdf8f3dd189980066ed0fd3df1fa59bfb54dd4ce51d7c44d7ef","observation_id":"7acb2adf-8935-4b29-b4b0-ced4b2456b51","resolution":{"observed_at":"2026-08-12T21:26:42.192559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.673285Z","title":"Extract- ing sae task features for in-context learning","venue":null,"work_id":"ec2fefba-a82c-4f21-8d04-75c7ee200667","year":null},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.197485Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:ceecd9209088362d7099494f0ebf0fd9d9f9019701a4848aa7c85be3f7a9dfe3","observation_id":"99106a46-df2c-4d72-b85a-c52220f51816","resolution":{"observed_at":"2026-08-12T21:26:42.678272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-14T07:41:23.271562Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-12T21:26:42.207022Z","title":"Gemma Scope: Open sparse autoencoders everywhere all at once on Gemma 2","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.207022Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:70dfa24a5e41eb445d42f6b3b5e1eab05c01e511f4f0bd850fe7d2896dee6b99","observation_id":"2a9da4f6-2dee-4565-876d-b723b2fe74df","resolution":{"observed_at":"2026-08-12T21:26:42.207022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06668","last_updated":"2024-02-13T22:37:39Z","snapshot_observed_at":"2026-08-13T20:57:38.293222Z","submitted_at":"2023-11-11T21:19:44Z","title":"In-context Vectors: Making In Context Learning More Effective and Controllable Through Latent Space Steering","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06668","snapshot_observed_at":"2026-08-12T21:26:42.212109Z","title":"In-context vectors: Making in context learning more effective and controllable through latent space steering","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.212109Z"},"links":{"cited_paper":"/paper/2311.06668","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:95dee5671af159927b3fee73afdf07e3ced41e8138ce24ecb3c93478069a05fe","observation_id":"ec9e69e6-7462-4872-b81b-c07493e66d53","resolution":{"observed_at":"2026-08-12T21:26:42.212109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19647","last_updated":"2025-03-27T05:44:45Z","snapshot_observed_at":"2026-08-14T07:35:01.333549Z","submitted_at":"2024-03-28T17:56:07Z","title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19647","snapshot_observed_at":"2026-08-12T21:26:42.218154Z","title":"Sparse feature circuits: Discovering and editing interpretable causal graphs in language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.218154Z"},"links":{"cited_paper":"/paper/2403.19647","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:53f2cb2c3a8fe98697af75ca0ae71ae1e322694e52efc41b2ec1151befb53fa9","observation_id":"5dcede98-43ca-41d7-84a4-ed6c5d787e5d","resolution":{"observed_at":"2026-08-12T21:26:42.218154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12522","last_updated":"2024-05-21T06:26:10Z","snapshot_observed_at":"2026-08-13T00:03:17.357784Z","submitted_at":"2024-05-21T06:26:10Z","title":"Sparse Autoencoders Enable Scalable and Reliable Circuit Identification in Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12522","snapshot_observed_at":"2026-08-12T21:26:42.223217Z","title":"Sparse autoencoders enable scalable and reliable circuit identification in language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.223217Z"},"links":{"cited_paper":"/paper/2405.12522","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:6ecca964c069d8d888f43cc6e567b23ebbc71af84d5b8dc34c8c61ac48f95826","observation_id":"250e0322-6fb2-4e22-9062-897dfbbd4407","resolution":{"observed_at":"2026-08-12T21:26:42.223217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06681","last_updated":"2024-07-05T15:30:45Z","snapshot_observed_at":"2026-08-07T14:28:06.805424Z","submitted_at":"2023-12-09T04:40:46Z","title":"Steering Llama 2 via Contrastive Activation Addition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06681","snapshot_observed_at":"2026-08-12T21:26:42.228251Z","title":"Steering llama 2 via contrastive activation addition","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.228251Z"},"links":{"cited_paper":"/paper/2312.06681","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:9eb98ccb89f43a46b0b140ad57382706c9623e3a877cd0ecf8709b3fcb5b27b5","observation_id":"858c7074-c469-4184-a78d-05176e644844","resolution":{"observed_at":"2026-08-12T21:26:42.228251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.09251","last_updated":"2022-12-19T05:13:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-19T05:13:52Z","title":"Discovering Language Model Behaviors with Model-Written Evaluations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.09251","snapshot_observed_at":"2026-08-12T21:26:42.233435Z","title":"Discovering language model behaviors with model-written evaluations","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.233435Z"},"links":{"cited_paper":"/paper/2212.09251","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:5c0953e95bba91af2aa8755a66633914e01bce471b8a75b3c2fdb90af7d38f13","observation_id":"cbd93b3a-c248-4dd1-a11f-8dfe341ab134","resolution":{"observed_at":"2026-08-12T21:26:42.233435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14435","last_updated":"2024-08-01T17:42:04Z","snapshot_observed_at":"2026-08-08T18:10:15.134088Z","submitted_at":"2024-07-19T16:07:19Z","title":"Jumping Ahead: Improving Reconstruction Fidelity with JumpReLU Sparse Autoencoders","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14435","snapshot_observed_at":"2026-08-12T21:26:42.238427Z","title":"Jumping ahead: Improving reconstruction fidelity with jumprelu sparse autoencoders","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.238427Z"},"links":{"cited_paper":"/paper/2407.14435","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:fd2282e0c35aa7ed0b28e782f01fe732bb225a3db6aa72493772b3168f16cd0b","observation_id":"aef3e9f5-2e52-4bf2-a019-fb1725854d4d","resolution":{"observed_at":"2026-08-12T21:26:42.238427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.640768Z","title":"Progress update #1 from the gdm mech interp team","venue":null,"work_id":"b4f4927e-5cca-4d0b-951a-706bf990a153","year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.243162Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:4d0ee08748a87f35c64efbe554fbb3c4fd21f5193ea7061d4272f64ac1b582ce","observation_id":"4f28daf5-ece9-4779-9f3b-0dcfc11a073e","resolution":{"observed_at":"2026-08-12T21:26:42.646166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.624569Z","title":"Steering vectors github, 2024","venue":null,"work_id":"fec0094d-0815-418d-8f54-dba660fda1a6","year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.247751Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:ac2dfa60103aea38ec675859aca5bbab865b1f46fb75138629f7bd1a8a5136b3","observation_id":"cee9464a-f715-447f-8ee0-9968495d0161","resolution":{"observed_at":"2026-08-12T21:26:42.630077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12404","last_updated":"2025-05-04T23:07:56Z","snapshot_observed_at":"2026-08-12T23:20:31.291372Z","submitted_at":"2024-07-17T08:32:03Z","title":"Analyzing the Generalization and Reliability of Steering Vectors","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12404","snapshot_observed_at":"2026-08-12T21:26:42.252157Z","title":"Analyzing the generalization and reliability of steering vectors","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.252157Z"},"links":{"cited_paper":"/paper/2407.12404","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:9216bb5f3821c41fbdbc07d397668c4e1555355fea9fab41ee8071e5a83b35f5","observation_id":"926a9237-9662-46cd-8027-eb0cfc23c41c","resolution":{"observed_at":"2026-08-12T21:26:42.252157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.609002Z","title":"Scaling monosemanticity: Extracting interpretable features from claude 3 sonnet","venue":null,"work_id":"bc046c89-2d7b-4493-b0c8-98d0217f51f2","year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.257201Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:06e9ba4f924cddf544176fff30f46b52a3ac8f22f52634be60cb89d3f940fa30","observation_id":"1d2a6ff2-a00b-4eb4-9098-40e0958cd4ff","resolution":{"observed_at":"2026-08-12T21:26:42.614375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.590078Z","title":"Vazquez, Ulisse Mini, and Monte MacDiarmid","venue":null,"work_id":"d20d0cfb-601d-4e41-af1e-d571b708a67d","year":null},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.262056Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:046fc84bad12449c49f03b757cf645ef61f278bc8e19b421041cff1f99fbd799","observation_id":"aa9d9698-8add-4c34-8774-58ada3ff1f26","resolution":{"observed_at":"2026-08-12T21:26:42.597872Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.13967","last_updated":"2025-03-01T01:35:47Z","snapshot_observed_at":"2026-08-13T00:01:51.398744Z","submitted_at":"2024-05-22T20:08:48Z","title":"Model Editing as a Robust and Denoised variant of DPO: A Case Study on Toxicity","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.13967","snapshot_observed_at":"2026-08-12T21:26:42.271629Z","title":"Model editing as a robust and denoised variant of dpo: A case study on toxicity, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.271629Z"},"links":{"cited_paper":"/paper/2405.13967","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:de982d3ceb781a0cd42c2bc0faa4a4a0a11fcf90d6f3ecf3220c8abe429d261e","observation_id":"5d45b20a-0088-4159-adf1-1876ebabf847","resolution":{"observed_at":"2026-08-12T21:26:42.271629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10248","last_updated":"2024-10-10T13:20:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-20T12:21:05Z","title":"Steering Language Models With Activation Engineering","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10248","snapshot_observed_at":"2026-08-12T21:26:42.266596Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.266596Z"},"links":{"cited_paper":"/paper/2308.10248","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:ddb146f3875b4d9b53cc14e21666d87a6fc647190cab6c2add06bc9421588ce6","observation_id":"cfc2846d-b426-4898-b074-e6399d138de9","resolution":{"observed_at":"2026-08-12T21:26:42.266596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-12T21:26:42.281399Z","title":"(A)” and “(B)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.281399Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:659032fe44bbae1201ddaa50ffb1cec42ce592009739cab5eabe243ab0a310a1","observation_id":"5e66b072-cd05-466f-a240-4a78e333966e","resolution":{"observed_at":"2026-08-12T21:26:42.281399Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05767","last_updated":"2024-03-09T02:30:04Z","snapshot_observed_at":"2026-08-13T00:58:54.709320Z","submitted_at":"2024-03-09T02:30:04Z","title":"Extending Activation Steering to Broad Skills and Multiple Behaviours","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05767","snapshot_observed_at":"2026-08-12T21:26:42.276584Z","title":"Extending activation steering to broad skills and multiple behaviours","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.276584Z"},"links":{"cited_paper":"/paper/2403.05767","citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:d8efda086858d281e1a6dd84c262ab8fe5b5a61408d1f5802ff1080d6d9ba14f","observation_id":"5582750b-97be-46e4-916b-82418641a8ce","resolution":{"observed_at":"2026-08-12T21:26:42.276584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T21:26:42.658024Z","title":null,"venue":null,"work_id":"c32d92a3-c1fb-47f8-8b68-87df8f6959c6","year":null},"citing_paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T21:26:42.202242Z"},"links":{"citing_paper":"/paper/2411.08790"},"observation_digest":"sha256:2528499b61a6b7a74146babc10f4cd347346ca98208452245bd495f9aaf00f95","observation_id":"e666d101-a9dd-4268-be4a-df728059110b","resolution":{"observed_at":"2026-08-12T21:26:42.662573Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.08790","last_updated":"2024-11-13T17:16:48Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-14T07:35:39.938773Z","submitted_at":"2024-11-13T17:16:48Z","title":"Can sparse autoencoders be used to decompose and interpret steering vectors?"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":18,"verified_exact":0,"verified_fuzzy":6},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 1 inbound Pith citation observation for arXiv:2411.08790."}