{"as_of":"2026-08-17T22:08:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:35045f635b404aeb2e635a874094d786428dffda304cfc61033ff20af53ded46","coverage":[{"denominator":105,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:25:06.108473Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T02:42:26.173782Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"cited_work":{"arxiv_id":"2505.01372","doi":"10.48550/arxiv.2505.01372","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.01372","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating","venue":"ArXiv.org","work_id":"f4a6a168-76c4-4b24-8d89-8609af3aa4e7","year":2025},"citing_paper":{"arxiv_id":"2605.08934","last_updated":"2026-06-16T20:17:16Z","snapshot_observed_at":"2026-08-11T08:05:03.537052Z","submitted_at":"2026-05-09T13:08:07Z","title":"From Mechanistic to Compositional Interpretability","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-12T02:42:26.173782Z"},"links":{"cited_paper":"/paper/2505.01372","citing_paper":"/paper/2605.08934"},"observation_digest":"sha256:98a737ab408ba623b4f6e61b84bae3291b5c566c6b12a95d117a65a77ca285bd","observation_id":"c7a68c25-a842-4499-bca8-0cea977ccc1b","resolution":{"observed_at":"2026-05-12T02:46:18.444788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.01372/citation-record","integrity":"/paper/2505.01372/integrity","json":"/paper/2505.01372/citation-record.json","paper":"/paper/2505.01372"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.08025","last_updated":"2025-04-01T14:16:47Z","snapshot_observed_at":"2026-08-17T18:31:03.354107Z","submitted_at":"2024-10-10T15:22:48Z","title":"The Computational Complexity of Circuit Discovery for Inner Interpretability","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08025","snapshot_observed_at":"2026-08-16T04:25:05.518727Z","title":"The computational complexity of circuit discovery for inner interpretability","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.518727Z"},"links":{"cited_paper":"/paper/2410.08025","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:01f8e28fc082a568a616d6d50f68dc2ff3702116f4c523601317a7d8cad319dd","observation_id":"7be2a119-13da-4a7a-a1d7-649e3984d9d1","resolution":{"observed_at":"2026-08-16T04:25:05.518727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14316","last_updated":"2024-07-16T10:22:51Z","snapshot_observed_at":"2026-08-16T14:57:46.371369Z","submitted_at":"2023-09-25T17:37:20Z","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14316","snapshot_observed_at":"2026-08-16T04:25:05.526366Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.526366Z"},"links":{"cited_paper":"/paper/2309.14316","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3a4a4f394251ce1cce0212426cb49af0c56296ef1f03888c1d2037e366b9b2b8","observation_id":"e96be6eb-3233-415f-bede-dba250dc41a7","resolution":{"observed_at":"2026-08-16T04:25:05.526366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.531896Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.531896Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:641adb544d2a476f1b6fa358abdeb1ea5c862d10afb48b3beffa81c25d44e5a8","observation_id":"61d387c8-950c-4507-82b2-d3eb05f6a283","resolution":{"observed_at":"2026-08-16T04:25:05.531896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.538644Z","title":"The urgency of interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.538644Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:faa7e840b57a889ee05e597f82fc2a9c2a3c0ef020fb54d403e0b73f81474628","observation_id":"aae00ee6-8ee0-419b-adff-b23b52bf598b","resolution":{"observed_at":"2026-08-16T04:25:05.538644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-08-16T14:00:45.842560Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-16T04:25:05.544723Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.544723Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:2f71a73780357f7aa78c96d1180d45c1bea0037b2d1a5281fdcae7e51e469c2f","observation_id":"deff4b33-b745-4e3a-9e7a-86e1815bb557","resolution":{"observed_at":"2026-08-16T04:25:05.544723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.550537Z","title":"Ai as systems, not just models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.550537Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:45ef746642145332389263d2f9a8c5e77c951eefc0f28ae0a93fab0a791aba48","observation_id":"964da59b-029d-4588-8631-c9b228d56532","resolution":{"observed_at":"2026-08-16T04:25:05.550537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.556998Z","title":"Standard saes might be incoherent: A choosing problem & a “concise” solution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.556998Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0b83858a83864b76f132ae5ffac9cbde7a45f3bb12d87d8af32580d7fe152ebd","observation_id":"7c179344-ca3a-48d8-82d1-3ba93e274d41","resolution":{"observed_at":"2026-08-16T04:25:05.556998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.563027Z","title":"Position: Interpretability is a bidirectional communication problem","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.563027Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d1d138e7721af9a4c824e0bcc2827b69a0785edd1776a69e017f16fdb6a6673e","observation_id":"1000c22c-0b9a-4e30-8e2f-592afc155ec8","resolution":{"observed_at":"2026-08-16T04:25:05.563027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.569755Z","title":"A mathematical philosophy of explanations in mechanistic interpretability: The strange science part i.i, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.569755Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d7db5d86eb30a07f8e1ba9b8e24b8089c2689f2e416bcae3f5b0acc0a33e30e3","observation_id":"237b7281-268c-41a9-b03e-66dee8e8639d","resolution":{"observed_at":"2026-08-16T04:25:05.569755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11179","last_updated":"2024-10-15T01:38:03Z","snapshot_observed_at":"2026-08-16T13:09:22.337092Z","submitted_at":"2024-10-15T01:38:03Z","title":"Interpretability as Compression: Reconsidering SAE Explanations of Neural Activations with MDL-SAEs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11179","snapshot_observed_at":"2026-08-16T04:25:05.575857Z","title":"Pearce, and Lee Sharkey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.575857Z"},"links":{"cited_paper":"/paper/2410.11179","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:607fe191a5d8e777b034b1e10150afe0a384d7a520041164b1a500ce5e0f4e0c","observation_id":"2f798255-8dd9-4727-8730-208b6e6b64e5","resolution":{"observed_at":"2026-08-16T04:25:05.575857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.581703Z","title":"Novum Organum","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.581703Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:774d314e582188ec5f535f7f5f16fe0132de61c9b536b7220d85fb5a82c74ed2","observation_id":"e326204e-4a5b-4139-967e-9d642129ebdf","resolution":{"observed_at":"2026-08-16T04:25:05.581703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.586827Z","title":"Simplicity","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.586827Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d847358dfc2e8e81973be05b3b8c0f1ffa1830571ae2af8e56f90e3d2e391b78","observation_id":"221e6ee1-3828-4e38-ba66-1b3543b46e24","resolution":{"observed_at":"2026-08-16T04:25:05.586827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.592337Z","title":"Design Rules: The Power of Modularity Volume 1","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.592337Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c5f65c24906be683f8ae8b49b73d7c9857b23f34909129b4481813a59f14ffdd","observation_id":"becd414b-4fff-4ff3-954f-598d5526b840","resolution":{"observed_at":"2026-08-16T04:25:05.592337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.597081Z","title":"Icml 2024 mechanistic interpretability workshop, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.597081Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0d842f26a48182a6c3507378025206c537df10e69f937b092f3e76b3aaff3dbc","observation_id":"98a5edf3-3bec-4513-b1e0-5c1ecb7f32cc","resolution":{"observed_at":"2026-08-16T04:25:05.597081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02981","last_updated":"2024-06-07T08:44:52Z","snapshot_observed_at":"2026-08-16T13:46:00.907207Z","submitted_at":"2024-06-05T06:23:49Z","title":"Local vs. Global Interpretability: A Computational Complexity Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02981","snapshot_observed_at":"2026-08-16T04:25:05.601776Z","title":"Local vs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.601776Z"},"links":{"cited_paper":"/paper/2406.02981","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6ae6870ed67d8a55c86df986364b6f68f4312c5a229d7d11f105b9f509032e67","observation_id":"64e4beca-6ae9-4858-b525-195cbebf0549","resolution":{"observed_at":"2026-08-16T04:25:05.601776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.shpsc.2005.03.010","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.256151Z","title":"Explanation: A mechanist alternative","venue":null,"work_id":"2a6b7a91-6a28-4964-94fc-7bde02b791c4","year":2005},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.606747Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:400b83238902ac9c7331df7cd8bbd19cc016471c9af86a45234b05610d89207d","observation_id":"3fa773a3-d922-4b64-81cd-e726fb950bf8","resolution":{"observed_at":"2026-08-16T04:25:06.262246Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.612261Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.612261Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:67f6e455f9fe11a44fcb20e370d1295fb139a4cb5d3af5f8dc6f95d5d78b1eb9","observation_id":"3c2e7bee-d2fd-4024-bc7c-1cb2370aa993","resolution":{"observed_at":"2026-08-16T04:25:05.612261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17805","last_updated":"2025-01-29T17:47:36Z","snapshot_observed_at":"2026-08-15T14:15:35.328292Z","submitted_at":"2025-01-29T17:47:36Z","title":"International AI Safety Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17805","snapshot_observed_at":"2026-08-16T04:25:05.617272Z","title":"International ai safety report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.617272Z"},"links":{"cited_paper":"/paper/2501.17805","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:2a5e9fe9a136b20c82f06c1a00867d56ad32a5f66c00fd2b3030651059430a86","observation_id":"c90eb841-85a5-44f5-b15d-ee4d768cc0c8","resolution":{"observed_at":"2026-08-16T04:25:05.617272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14082","last_updated":"2024-08-23T23:02:28Z","snapshot_observed_at":"2026-07-06T18:03:38.397804Z","submitted_at":"2024-04-22T11:01:51Z","title":"Mechanistic Interpretability for AI Safety -- A Review","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14082","snapshot_observed_at":"2026-08-16T04:25:05.622447Z","title":"Mechanistic Interpretability for AI Safety -- A Review , April 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.622447Z"},"links":{"cited_paper":"/paper/2404.14082","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6e00663e963cac7591e16cc3b5080730be6485f5a300622aff57d2faee0dfa70","observation_id":"ec496079-ff91-465c-b9db-71c5d491d4cc","resolution":{"observed_at":"2026-08-16T04:25:05.622447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2307/2180883","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.233516Z","title":null,"venue":null,"work_id":"23c0d7b2-0ec4-4d02-a0b0-039256e91932","year":1940},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.627697Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:bcc26c0684611d0b592dff5e75808deecf7897616a2ce839b98fb49df7d00b00","observation_id":"03f4349d-828d-465b-9065-1c382d45670d","resolution":{"observed_at":"2026-08-16T04:25:06.239659Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.632693Z","title":"Auditing local explanations is hard","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.632693Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:01a7637ac337a9c8dd35e76c502b1deff83764ece030c9469c244ead11b7d04b","observation_id":"c960ebea-2022-4f48-a643-0b57db62aa68","resolution":{"observed_at":"2026-08-16T04:25:05.632693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.637441Z","title":"Language models can explain neurons in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.637441Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:85b6262015370668e268450353e6176414037347c62d4ec609bcffb8f2dfda0a","observation_id":"ed2767c7-3100-4b87-beb9-87dce637ba3a","resolution":{"observed_at":"2026-08-16T04:25:05.637441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.07143","last_updated":"2021-04-14T22:04:48Z","snapshot_observed_at":"2026-08-17T09:17:27.672646Z","submitted_at":"2021-04-14T22:04:48Z","title":"An Interpretability Illusion for BERT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.07143","snapshot_observed_at":"2026-08-16T04:25:05.642287Z","title":"An Interpretability Illusion for BERT , April 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.642287Z"},"links":{"cited_paper":"/paper/2104.07143","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3b5ea6d01e942d66f7dab8ca80f6be19e234b637cd9ab5e538afae3e00ab26e9","observation_id":"87cc22af-9945-4683-b20e-ef73fe38ec6a","resolution":{"observed_at":"2026-08-16T04:25:05.642287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12241","last_updated":"2024-05-24T13:16:32Z","snapshot_observed_at":"2026-08-16T13:51:48.413485Z","submitted_at":"2024-05-17T17:03:46Z","title":"Identifying Functionally Important Features with End-to-End Sparse Dictionary Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12241","snapshot_observed_at":"2026-08-16T04:25:05.647126Z","title":"Identifying Functionally Important Features with End -to- End Sparse Dictionary Learning , May 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.647126Z"},"links":{"cited_paper":"/paper/2405.12241","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a3c2979f8a2307c10a5263e61378e0cb978bf18924c406bae85478386b2c10f3","observation_id":"a2b41a39-4f83-4cfd-a9f0-3663c232c3da","resolution":{"observed_at":"2026-08-16T04:25:05.647126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14926","last_updated":"2025-02-07T19:22:32Z","snapshot_observed_at":"2026-08-14T20:11:56.886595Z","submitted_at":"2025-01-24T21:31:12Z","title":"Interpretability in Parameter Space: Minimizing Mechanistic Description Length with Attribution-based Parameter Decomposition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14926","snapshot_observed_at":"2026-08-16T04:25:05.652281Z","title":"Interpretability in parameter space: Minimizing mechanistic description length with attribution-based parameter decomposition","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.652281Z"},"links":{"cited_paper":"/paper/2501.14926","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:445f8ac2c75219c805301bd3dcbb4301776029a38de68132260b01c045730962","observation_id":"3a76e277-6588-436b-b83d-5be24a5b09ec","resolution":{"observed_at":"2026-08-16T04:25:05.652281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.658022Z","title":"Towards Monosemanticity : Decomposing Language Models With Dictionary Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.658022Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:93ccc627bdf5e25b8836fc85781a4884c634c7edbeaf0e11c5430b8074aaacd9","observation_id":"135608c9-8774-44fb-bf3e-18692e656f90","resolution":{"observed_at":"2026-08-16T04:25:05.658022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15740","last_updated":"2025-01-27T03:06:06Z","snapshot_observed_at":"2026-08-13T19:26:18.474749Z","submitted_at":"2025-01-27T03:06:06Z","title":"Propositional Interpretability in Artificial Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15740","snapshot_observed_at":"2026-08-16T04:25:05.663766Z","title":"Propositional interpretability in artificial intelligence","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.663766Z"},"links":{"cited_paper":"/paper/2501.15740","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f2e097f4b8bbd3c5781cdbac3f3d2fd02012424d2ae99ac893e33d197e46b6fc","observation_id":"34747481-f21c-4452-9e82-01aac17a063e","resolution":{"observed_at":"2026-08-16T04:25:05.663766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.669143Z","title":"A toy model of universality: Reverse engineering how networks learn group operations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.669143Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:88b6033da4a3c9c49de420fb4be96e61a80f1f756c3390d38e944947dcf0f32c","observation_id":"8c79eff4-85ec-4984-8618-b2188376c0c3","resolution":{"observed_at":"2026-08-16T04:25:05.669143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.674171Z","title":"The evolutionary origins of modularity","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.674171Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ffacffdd402bb330347e24d9780181d8dcb01b074d088940f8fc84aabde39e4a","observation_id":"ab6e7c15-147c-49e3-a67e-2ee2331be54d","resolution":{"observed_at":"2026-08-16T04:25:05.674171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14997","last_updated":"2023-10-28T20:05:52Z","snapshot_observed_at":"2026-08-16T15:36:54.698423Z","submitted_at":"2023-04-28T17:36:53Z","title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14997","snapshot_observed_at":"2026-08-16T04:25:05.678871Z","title":"Mavor-Parker, Aengus Lynch, Stefan Heimersheim, and Adrià Garriga-Alonso","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.678871Z"},"links":{"cited_paper":"/paper/2304.14997","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a5e0f44c54329c052aaf7886485390ebaaddff3ee99eab2de77298ea5f0bfef2","observation_id":"0bbc4ffc-89aa-4a8a-9228-f21bfc00c894","resolution":{"observed_at":"2026-08-16T04:25:05.678871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.684525Z","title":"Central dogma of molecular biology","venue":null,"work_id":null,"year":1970},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.684525Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:5feda80ece456a97a20d64950593558324b08213c4ec74c9b3023b6901703939","observation_id":"288ced46-da29-40db-ae02-4244fd08479c","resolution":{"observed_at":"2026-08-16T04:25:05.684525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.690760Z","title":"The beginning of infinity: Explanations that transform the world","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.690760Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:46f01f45c0feb771647da1ade591131b4491c140992bdc74a6129a58ae3daccc","observation_id":"0076b1cc-574b-4515-bbd6-8b68134604ec","resolution":{"observed_at":"2026-08-16T04:25:05.690760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.696683Z","title":null,"venue":null,"work_id":null,"year":1919},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.696683Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:e959bb0f63e8694d71d77bd1db082e95ea56626d9d0a9aa7d0c77a45319d95fb","observation_id":"9e2e0718-96c4-4dee-82a6-24b4caa1f973","resolution":{"observed_at":"2026-08-16T04:25:05.696683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.702151Z","title":"Einstein","venue":null,"work_id":null,"year":1916},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.702151Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d2e59ae07d8526aa0b1bf41f4be0d51eff61cd599591fb76a02f05d8aec009d7","observation_id":"1643f022-9fd7-4021-af3f-81f37f4af433","resolution":{"observed_at":"2026-08-16T04:25:05.702151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.707415Z","title":null,"venue":null,"work_id":null,"year":1974},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.707415Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3fb68bdc15225374af45963713daf75e30c07b420b85701431f6fbd33e9c1ada","observation_id":"6badf7f6-0bd7-48dd-b6d1-dbc160e7f37b","resolution":{"observed_at":"2026-08-16T04:25:05.707415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03386","last_updated":"2021-03-04T23:53:53Z","snapshot_observed_at":"2026-08-16T18:41:07.307930Z","submitted_at":"2021-03-04T23:53:53Z","title":"Clusterability in Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03386","snapshot_observed_at":"2026-08-16T04:25:05.712412Z","title":"Clusterability in neural networks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.712412Z"},"links":{"cited_paper":"/paper/2103.03386","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a0dbc0449a5362aed7e7bc9876d28c2d55ec2aac9c8a1226294a46bb5820dc21","observation_id":"465f48a9-16d1-4854-9556-2b1edce37af5","resolution":{"observed_at":"2026-08-16T04:25:05.712412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.718000Z","title":"Interpretability illusions in the generalization of simplified models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.718000Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:85c187fa769d38503ac28bccef2c6a0adec9405210f3748ce1243ee5fca69a27","observation_id":"d72172f6-6ee7-4cf2-a87e-c8aa458a6a2e","resolution":{"observed_at":"2026-08-16T04:25:05.718000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04093","snapshot_observed_at":"2026-08-16T04:25:05.723010Z","title":"Scaling and evaluating sparse autoencoders, June 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.723010Z"},"links":{"cited_paper":"/paper/2406.04093","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1a7580796f858eaaf222f77580edd868702b25407d95940bfa15ba6460e1d94e","observation_id":"f97a349d-416e-4ec6-ba95-63ed4aa9bd7f","resolution":{"observed_at":"2026-08-16T04:25:05.723010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.728174Z","title":"Causal abstractions of neural networks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.728174Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d93c504d05379bb430ee46b4514ded8ea04093047dc5fbed765858794c120ef9","observation_id":"712695dc-797c-4577-a954-6c0f803ec29b","resolution":{"observed_at":"2026-08-16T04:25:05.728174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-16T04:25:05.733991Z","title":"Causal abstraction for faithful model interpretation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.733991Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ca170fde866d3490b2e236206f7807850a98dc38378e8b351695184b49d64fe8","observation_id":"22aea982-0987-44f7-ab5b-592ababe55dd","resolution":{"observed_at":"2026-08-16T04:25:05.733991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.740650Z","title":"Clustering algorithms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.740650Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:404aa653f060f850eeaa2484346c3812e0fe1dff04c6c00ef3b79adfad495f6a","observation_id":"25eafe53-c2b4-44c6-a494-fd448e1bf91c","resolution":{"observed_at":"2026-08-16T04:25:05.740650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.746673Z","title":"Compact proofs of model performance via mechanistic interpretability","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.746673Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d44274048e436cad4be3a6d1789b69cb22a6b404a5200918d4738d7af3686ac4","observation_id":"5d25dd86-a4ae-4bdc-bad7-135601377bcb","resolution":{"observed_at":"2026-08-16T04:25:05.746673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.870126Z","title":"Interpbench: Semi-synthetic transformers for evaluating mechanistic interpretability techniques","venue":null,"work_id":"e54e80b9-a160-4d8b-b2a4-7bf9d719c953","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.752590Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:76e5c17b4a51f1c855f10003fe549323dd6fa7a284b8925c5249ce342e1aacf8","observation_id":"33317061-ea4a-44d4-a1cc-f3220fce6bba","resolution":{"observed_at":"2026-08-16T04:25:07.875795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.851429Z","title":"Hastie, R","venue":null,"work_id":"10fa1ef4-4cd8-4c81-9b61-59137ecab1d4","year":2009},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.757293Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:71b3b97ceb123efeadf03fadd55f140a3c51a7a69a8afbeadcdc265c40f49596","observation_id":"6c1fbada-0a31-460d-8864-8ab9e751bc15","resolution":{"observed_at":"2026-08-16T04:25:07.856843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.830083Z","title":"Hempel and Paul Oppenheim","venue":null,"work_id":"f52e96cd-5c86-4140-aca9-b1954c633af1","year":1948},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.762261Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:64751099394f7c20ae809823a6ef5e34c6babacad2d2c82c1ea96e1162065a61","observation_id":"6bb4f5b1-e9b0-4de3-b18b-8a9922be0e2e","resolution":{"observed_at":"2026-08-16T04:25:07.837001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.811267Z","title":"Philosophy of Natural Science","venue":null,"work_id":"95ab8c89-9a9e-4ddd-877c-0eb1ce95b61a","year":1966},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.767579Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7f37581cc37e19a4ff6b950452eecd862537b5d642a953936e7b6479ee4fac1d","observation_id":"0fc0361d-3d60-41e5-9e5c-c8186b9233c4","resolution":{"observed_at":"2026-08-16T04:25:07.816805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.791761Z","title":"Bayesianism and inference to the best explanation","venue":null,"work_id":"99363bfb-e214-486a-925d-3bdc68758351","year":2014},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.773005Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ee1f205ca53d4b6bc054939dfba295e52987f67e2026fabb8dd85319ae623b60","observation_id":"e355e61b-deec-4081-8248-fe70d5984b8b","resolution":{"observed_at":"2026-08-16T04:25:07.797867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02364","last_updated":"2025-08-01T09:20:40Z","snapshot_observed_at":"2026-08-16T14:21:46.708146Z","submitted_at":"2024-02-04T06:23:05Z","title":"Loss Landscape Degeneracy and Stagewise Development in Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02364","snapshot_observed_at":"2026-08-16T04:25:05.779468Z","title":"The developmental landscape of in-context learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.779468Z"},"links":{"cited_paper":"/paper/2402.02364","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:5273493fa210666e8a7fff44a7313c3b87f6515fb36e4e84925388a2a890fca7","observation_id":"b0510d13-33cf-4cb9-bfb5-b1bc7a98b276","resolution":{"observed_at":"2026-08-16T04:25:05.779468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.785160Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.785160Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:66e138d2471746c2f9500f06e94573923c081d289953722c3082d01cca10cc22","observation_id":"c5ec30f4-92f1-40f8-9acf-a47b1637be70","resolution":{"observed_at":"2026-08-16T04:25:05.785160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.762063Z","title":"Hutter, E","venue":null,"work_id":"e762c19b-f9c8-4491-9749-ee475bc854f3","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.791367Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c70d271a8993277ee94b69db046ae573fc5c5235caf91923990ca237c0c33344","observation_id":"40da69cb-7913-46ec-9031-94e2f0f81322","resolution":{"observed_at":"2026-08-16T04:25:07.767475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.743207Z","title":"Fine-tuning neural networks to match their interpretation: Towards scaling compact proofs, 2025","venue":null,"work_id":"8877e86b-c974-4766-a0a8-7b338ad07725","year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.796480Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:970d639d44efe863b2bb4c99fc6468a1ba3e91b691b50a80eaf21d0fe007fcf7","observation_id":"1c3adc74-c6de-4fdf-b320-a17fec806d6d","resolution":{"observed_at":"2026-08-16T04:25:07.749269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.717970Z","title":"Kandel, J.H","venue":null,"work_id":"bb27f722-1eb1-4bf5-a0d2-7417f61e3b44","year":2000},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.801635Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:970e57ee85991cc04877c25f8ccbafa95eb0d7e0de828523f2ef35a488b6cdef","observation_id":"7ff9effe-ae4b-46a8-be2e-3ce3887cac71","resolution":{"observed_at":"2026-08-16T04:25:07.729487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.699617Z","title":"Saebench: a comprehensive benchmark for sparse autoencoders, 2024","venue":null,"work_id":"5a4c3359-9828-474a-9cf3-805cb797679e","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.807554Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b006f5e98af883773b8297607d685d6176b49e2083266d16f358c31997ddd7a6","observation_id":"6d585e48-eb4a-4bc0-9aac-d340d76ab5f0","resolution":{"observed_at":"2026-08-16T04:25:07.705111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.681692Z","title":"Kennefick","venue":null,"work_id":"da9e08bd-579b-4c35-b0a4-c3081af48560","year":1919},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.814962Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:366ac1860aff3ffbcea8cf891418560a4bc185e6d0751778ca82deb43982ee27","observation_id":"fd84da5c-f9e0-43b8-96b3-ed972dd73d61","resolution":{"observed_at":"2026-08-16T04:25:07.687071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.661190Z","title":"Explanatory unification","venue":null,"work_id":"9b1f8f18-7195-4bc6-9168-7fd53a46d205","year":1981},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.820191Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:4203be646db8eb75819510a6a5408002df7dc244f8a395f7cd322a496cfad4fd","observation_id":"16b197c7-a52f-41ad-bccc-66f48285f750","resolution":{"observed_at":"2026-08-16T04:25:07.667802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.639413Z","title":"Three approaches to the quantitative definition ofinformation","venue":null,"work_id":"3cd4f4b2-2ab8-4786-b625-46fd32db056c","year":1965},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.825015Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b5be66a8d81039c786e92d1ae5a0159e32fa9ea7f1dcb06211b736b772f907e6","observation_id":"10e71510-9e4a-4a0e-ad3c-0aaf7a4baba4","resolution":{"observed_at":"2026-08-16T04:25:07.646435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.622120Z","title":null,"venue":null,"work_id":"71934480-c0fb-49d2-9cd5-378330a1027e","year":1981},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.830341Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0d08650279770369860cbe676f9ca9350f2037553f0a0ccee3ef43c51a3293ca","observation_id":"17746ef1-137a-4c4f-b7fd-8d3fa293643b","resolution":{"observed_at":"2026-08-16T04:25:07.627680Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.598916Z","title":"The Structure of Scientific Revolutions","venue":null,"work_id":"14dd557c-c095-45d3-9e1e-d6135eed97c9","year":1962},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.835464Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:8c965ab5a3567c708dc154d513147237e1233bc044e861a9f7e171bf2972e04b","observation_id":"1587381a-db0b-4c29-8c54-0a0c152ca444","resolution":{"observed_at":"2026-08-16T04:25:07.607038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.580037Z","title":"Falsification and the methodology of scientific research programmes","venue":null,"work_id":"2331be14-3ed4-4689-a41c-f0a22bfb8b75","year":1970},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.840481Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1f94a35673fe7db954bb3a67404e0c6b0895b90f3d80b571b186d8184e8b6f58","observation_id":"aaa33f02-0ff5-4a53-aada-a5d9be8f6d4f","resolution":{"observed_at":"2026-08-16T04:25:07.586373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.562069Z","title":"The Methodology of Scientific Research Programmes","venue":null,"work_id":"13d98109-e392-4f1a-b37b-fd049ab94603","year":1978},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.845034Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:502a1310e13c302fe9ddc7653dc85cab289b1785582393c5024f7f979cb56062","observation_id":"91c964fc-0773-420c-8279-fc1eb53403b3","resolution":{"observed_at":"2026-08-16T04:25:07.567371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04878","last_updated":"2025-02-07T12:33:08Z","snapshot_observed_at":"2026-08-13T12:18:38.447222Z","submitted_at":"2025-02-07T12:33:08Z","title":"Sparse Autoencoders Do Not Find Canonical Units of Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04878","snapshot_observed_at":"2026-08-16T04:25:05.850864Z","title":"Sparse autoencoders do not find canonical units of analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.850864Z"},"links":{"cited_paper":"/paper/2502.04878","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:93bf982e0a314aed64f85d887e18955ad44f90f7267956e952d0b9ca7d64a155","observation_id":"f9885b31-5dd2-4dae-97f1-f880fb428f8b","resolution":{"observed_at":"2026-08-16T04:25:05.850864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.856242Z","title":"Lindsay and David Bau","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.856242Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ec024184858ba2c7dfc26695eb28cceec3b9564c0ca7db55948734ff957f21f2","observation_id":"f7a9f288-c896-4fba-9c44-d8753e20975f","resolution":{"observed_at":"2026-08-16T04:25:05.856242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.862843Z","title":"The mythos of model interpretability: In machine learning, the concept of interpretability is both important and slippery","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.862843Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:20eaa44f98459b619cb564268c6576dd942f0fe896ed0a9ba743cec65551cab5","observation_id":"b0ae1407-4a60-447c-8d2e-dada1f82ccfc","resolution":{"observed_at":"2026-08-16T04:25:05.862843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.533799Z","title":"Mechanistic mode connectivity","venue":null,"work_id":"26377831-d4e0-4b0d-b5cd-708f55fdbaac","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.868951Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:94c6f10ad17542c6683cb6a955de56c6a42ab480dd31d688ce1a23511f9c8230","observation_id":"9eb3975f-8e1f-4e66-a474-ec9c3b1a62af","resolution":{"observed_at":"2026-08-16T04:25:07.539180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.874084Z","title":"Information theory, inference and learning algorithms","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.874084Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:cdd68107a1568086ba383b946fdea97569f8b3cb8399a1d15a58cf87eb63a90f","observation_id":"ca10ed64-93ed-4ac4-a197-12840824058c","resolution":{"observed_at":"2026-08-16T04:25:05.874084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.501630Z","title":"Is this the subspace you are looking for? an interpretability illusion for subspace activation patching","venue":null,"work_id":"e7ed6346-bcf8-4546-8016-152a5e6a7e55","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.879925Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:649e3e5e543589578dc6812ae4f864b842de9d59e08e98b5adb485ace06bad2a","observation_id":"86793f78-8c9c-49a5-b6b4-0c5e352fa2df","resolution":{"observed_at":"2026-08-16T04:25:07.507881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.481672Z","title":"Downstream applications as validation of interpretability progress, March 2025","venue":null,"work_id":"7cc98727-fb6a-46cf-8d2c-f0eba287fccf","year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.885763Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:619c9e6a842cd95d939e420b7c3af952709c040d44ab8290e4283fb558e4111e","observation_id":"2085c2d2-ad9b-49d1-9c1b-412de83b2a98","resolution":{"observed_at":"2026-08-16T04:25:07.486729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19647","last_updated":"2025-03-27T05:44:45Z","snapshot_observed_at":"2026-08-14T07:35:01.333549Z","submitted_at":"2024-03-28T17:56:07Z","title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19647","snapshot_observed_at":"2026-08-16T04:25:05.891673Z","title":"Michaud, Yonatan Belinkov, David Bau, and Aaron Mueller","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.891673Z"},"links":{"cited_paper":"/paper/2403.19647","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a9133bceca3e04d2343ff81a1f6aa7cfd87af364f8d28f8e0fc4bd5339582118","observation_id":"484ea65f-aec7-463d-80ea-4d5c9dc4fd29","resolution":{"observed_at":"2026-08-16T04:25:05.891673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.463184Z","title":"Cognitive styles in two cognitive sciences","venue":null,"work_id":"a95f36a6-b00e-48b4-a249-8ba993ee1299","year":2012},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.898471Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7c0627ed6e62d64ac4106fd45b3c8112f96280ad83679dedf04c8f1cbef5dbc4","observation_id":"eb401795-6468-4d96-8e49-cfb9a832798b","resolution":{"observed_at":"2026-08-16T04:25:07.468750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.905678Z","title":"Zoom in: An introduction to circuits","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.905678Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:2cbe7984ca3f2981349bbdb2f8e0c5f8a63c4f3c82e79d4932d3f143eff77411","observation_id":"fb3e9be1-3e90-4402-bf7a-be0ded544915","resolution":{"observed_at":"2026-08-16T04:25:05.905678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-08-15T09:43:59.961298Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-16T04:25:05.911761Z","title":"In-context learning and induction heads","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.911761Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:31bcabbd76133ec7b0abfd7edaa73dad09dac5cf59a237e85a8031474aea64b3","observation_id":"536f7abc-d582-4db8-9057-fae24a5dba45","resolution":{"observed_at":"2026-08-16T04:25:05.911761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13928","last_updated":"2025-08-06T13:47:10Z","snapshot_observed_at":"2026-08-16T13:08:17.757544Z","submitted_at":"2024-10-17T17:56:01Z","title":"Automatically Interpreting Millions of Features in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13928","snapshot_observed_at":"2026-08-16T04:25:05.918441Z","title":"Automatically interpreting millions of features in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.918441Z"},"links":{"cited_paper":"/paper/2410.13928","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:69d186807c47028fa2183bbfd4142216da2d59737524bc56743a6ff6ebe3d453","observation_id":"293d1e50-1950-4e08-84f3-51ad6d5bb269","resolution":{"observed_at":"2026-08-16T04:25:05.918441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.925700Z","title":"Causality","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.925700Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6299a2e52d40ad9abe7388584267248546082640778441b9b28e15991cd60a6b","observation_id":"3b0b26c7-9cc4-4b96-8dd8-8f02df58426d","resolution":{"observed_at":"2026-08-16T04:25:05.925700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.410379Z","title":"Poincar \\'e","venue":null,"work_id":"caa93d46-3962-447f-afa1-4017648bbfaf","year":1905},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.932748Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:055205eed8a0810fbe84ff4abcea4181f009dc9897761c7824050fe601dd2de7","observation_id":"c6aebdff-6c3d-404e-b0fd-1062e6f53578","resolution":{"observed_at":"2026-08-16T04:25:07.416845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.944080Z","title":null,"venue":null,"work_id":null,"year":1935},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.944080Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:9dbbaa98efed3fe4c137feba421b5945f5ad09199cc8104ecbec5607e82a7388","observation_id":"e2af5c31-1409-4f9f-b0c2-e63b82fc5688","resolution":{"observed_at":"2026-08-16T04:25:05.944080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3998/phimp.1521","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.216159Z","title":"Hume on theoretical simplicity","venue":null,"work_id":"da413d8b-0960-4631-953f-4133a331a693","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.951779Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1c7dbe80d001eca7348acd483d560856feccea6565e6e856d66be4deb37adef4","observation_id":"5b60c837-b55c-4963-8912-cee876b825cf","resolution":{"observed_at":"2026-08-16T04:25:06.221515Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.962513Z","title":"Escalation risks from language models in military and diplomatic decision-making","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.962513Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:eb60515a88293ec5f232904b7e469f68724a357681b0085149974a7addda5094","observation_id":"e7de4f8c-c5f8-4c6e-bdb3-6d566e58bd11","resolution":{"observed_at":"2026-08-16T04:25:05.962513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.378688Z","title":"Four decades of scientific explanation","venue":null,"work_id":"fb44d7a8-d8f2-49fb-98a2-f178187b2846","year":1989},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.968237Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:bde09a220039cf5d924cabfb93ca4b17f8b370c69bce8c6bc3e64aa38eb6b261","observation_id":"4747c415-0ae3-4b04-b1ee-2029e9bd4a1f","resolution":{"observed_at":"2026-08-16T04:25:07.384390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.358969Z","title":null,"venue":null,"work_id":"9665ca3d-638b-4587-aede-53acc4fff13f","year":1984},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.973050Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6155324ba47a54f1f93ef0f4aa651734a83cbb5463986be400868093ebfd38d0","observation_id":"80943169-dfbc-466b-a923-efe22d9afad1","resolution":{"observed_at":"2026-08-16T04:25:07.365849Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09087","last_updated":"2024-10-07T15:02:12Z","snapshot_observed_at":"2026-08-16T13:11:52.807313Z","submitted_at":"2024-10-07T15:02:12Z","title":"Mechanistic?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09087","snapshot_observed_at":"2026-08-16T04:25:05.978424Z","title":"Mechanistic?, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.978424Z"},"links":{"cited_paper":"/paper/2410.09087","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6493086c5ba997339680160ed3ef2eac159fa07c8e58b827663e25742f9b8249","observation_id":"29c5b0f6-51c4-4735-99a3-2c9f05b2f048","resolution":{"observed_at":"2026-08-16T04:25:05.978424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.339909Z","title":"Theoretical Virtues in Science: Uncovering Reality Through Theory","venue":null,"work_id":"6f56e31d-b057-47a8-a0b0-781e12bf3633","year":2018},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.984022Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7a4e5835517b3ffbfb956ab5b7dc1fd3690e0f72dc1982eb2cfef27e05d7a90c","observation_id":"08075a01-c91b-4cc8-8e58-12d0004ac73c","resolution":{"observed_at":"2026-08-16T04:25:07.346087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.990103Z","title":"Riechers, Lucas Teixeira, Alexander Gietelink Oldenziel, and Sarah Marzen","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.990103Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6309f11a716b6a832365af4da62ebb00e4c57c37f41101372dc76d3633c7bdf7","observation_id":"740ce7be-8b7a-4b84-b752-dd27ef2f759c","resolution":{"observed_at":"2026-08-16T04:25:05.990103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.996982Z","title":"A mathematical theory of communication","venue":null,"work_id":null,"year":1948},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.996982Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7a518937a8b116b94243e0607446687266a1588d1b24f1fc4e58d958701dd5b6","observation_id":"1b7789d5-9c8b-4ace-9512-ac09ab787734","resolution":{"observed_at":"2026-08-16T04:25:05.996982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16496","last_updated":"2025-01-27T20:57:18Z","snapshot_observed_at":"2026-07-06T20:27:04.664873Z","submitted_at":"2025-01-27T20:57:18Z","title":"Open Problems in Mechanistic Interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16496","snapshot_observed_at":"2026-08-16T04:25:06.003123Z","title":"Michaud, Stephen Casper, Max Tegmark, William Saunders, David Bau, Eric Todd, Atticus Geiger, Mor Geva, Jesse Hoogland, Daniel Murfet, and Tom McGrath","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.003123Z"},"links":{"cited_paper":"/paper/2501.16496","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:68193045e2775732fafc500a40ababce3feb5beac63474b86f0aa2eaa0811ebf","observation_id":"b417d3b7-a3fc-4a92-a138-97465a1fe086","resolution":{"observed_at":"2026-08-16T04:25:06.003123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.290004Z","title":"Hypothesis testing the circuit hypothesis in LLM s","venue":null,"work_id":"ec2f18a9-e756-41d2-8ad4-0a2ae0a5bd50","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.010034Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d83742bffb80ea7fe3719aa411988cc1767337ab6687ca4b40b2c2bdbf4a8395","observation_id":"e07c0ebc-f326-427a-a03f-bbc9cbaf8612","resolution":{"observed_at":"2026-08-16T04:25:07.296737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.272093Z","title":"The golden mean of scientific virtues, 2024","venue":null,"work_id":"ee1a0dc9-ed96-40ab-9eda-eb57113de866","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.016375Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7ccd5e71315efe4b600790cc954ee3a158fe37f669ebf1cd0c22d2d03891ed98","observation_id":"1d512897-3ac1-4f50-b804-40995ead7c76","resolution":{"observed_at":"2026-08-16T04:25:07.276920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.253076Z","title":"Knowledge in Perspective: Selected Essays in Epistemology","venue":null,"work_id":"5548e441-35e6-4b62-a6c5-029f65b7448c","year":1991},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.024761Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b93eb39e5c4e94df6b5fbd8b32ed57296245a2757ae3c03f65deb49df9d176f7","observation_id":"1d7ce50c-4f21-4f8c-a2d0-82bfdc5f1ff6","resolution":{"observed_at":"2026-08-16T04:25:07.259403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.233793Z","title":"Grokking group multiplication with cosets","venue":null,"work_id":"2fe8047a-1207-4fd2-b883-1e4d7b92568c","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.031559Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f4a895ef610009d3e68e1d24372cd1187f4b1dc78f230f38d9cb7dedf00ff70a","observation_id":"352006f4-c9b9-4bf7-bc3c-71372f66f6b8","resolution":{"observed_at":"2026-08-16T04:25:07.240232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.212737Z","title":"Simplicity as Evidence of Truth","venue":null,"work_id":"65c882b4-dcf0-4f1b-919c-b43e6aa19d5a","year":1997},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.037028Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:dbb40201edd3d4228d743c2ec4f400559a709227daff1fdc4080e81021ba150c","observation_id":"ca0eab37-098d-4e05-a20d-024b68c63bb8","resolution":{"observed_at":"2026-08-16T04:25:07.218842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.13714","last_updated":"2024-09-07T10:02:51Z","snapshot_observed_at":"2026-08-16T13:20:33.519678Z","submitted_at":"2024-09-07T10:02:51Z","title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","version":1},"cited_work":{"arxiv_id":"2409.13714","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.13714","snapshot_observed_at":"2026-08-16T04:25:06.323657Z","title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","venue":"cs.CL","work_id":"24f26777-fae9-4acf-a74a-3367b6104d79","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.043348Z"},"links":{"cited_paper":"/paper/2409.13714","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:104ef251074987b232e6ce0d4d5568a9032fa1501257ddb872e2a70ed73ec3e0","observation_id":"2e2cb470-0087-4b83-a9c2-b26796256c1f","resolution":{"observed_at":"2026-08-16T04:25:06.330001Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.185289Z","title":"Interpretability in the wild: a circuit for indirect object identification in gpt-2 small","venue":null,"work_id":"01114bac-7a99-4508-abf8-22ec661a5a7a","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.049199Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b1564dc38ca3e114e8734cb110de549d09c3fb1ae4aed23a729ec38bbab995aa","observation_id":"0a2c7ecf-696a-4a1a-9927-dc96616081c7","resolution":{"observed_at":"2026-08-16T04:25:07.193035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1093/analys/65.3.205","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.194033Z","title":"Why favour simplicity? Analysis, 65 0 (3): 0 205--210, 2005","venue":null,"work_id":"7330de9a-0f1b-49d5-a62c-3a6f6cdbd97e","year":2005},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.057524Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f73f34db997e9d72af58d53e3d28a270f9b9a49fa87a1406261df2b11767e7d5","observation_id":"57fadde9-0811-439f-84ad-fa1f01846649","resolution":{"observed_at":"2026-08-16T04:25:06.202634Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.167522Z","title":"Understanding as compression","venue":null,"work_id":"74dbc341-e9f4-4760-833f-240ee97d2ba4","year":2019},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.064714Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3045accbc63d1463fbbf4bd287101bd0140039b4df05a40fbfdbf323f2ecaadb","observation_id":"b37e3d23-7ad3-4fdd-b774-1a560c33b6d1","resolution":{"observed_at":"2026-08-16T04:25:07.173192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.148654Z","title":"Geschichte und Naturwissenschaft","venue":null,"work_id":"c70f713d-4f40-4cc9-ac2f-e023b2b492aa","year":null},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.071370Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:710d934a8fd6fcf0bbd6bcdbaa9862618ce8590bdc7606d6ad70378cce7b69e3","observation_id":"095fdfb3-67ab-48ae-a712-b6d6f312a35d","resolution":{"observed_at":"2026-08-16T04:25:07.154724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.129771Z","title":"From probability to consilience: How explanatory values implement bayesian reasoning","venue":null,"work_id":"e58874bb-1f35-454d-8f79-af96acc32162","year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.076238Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:30234f68b9451e3b61715ff0f18801d76648567014ccd3d04485df82551fb063","observation_id":"721adab6-90b7-48eb-81f9-c03e9c9c1263","resolution":{"observed_at":"2026-08-16T04:25:07.136110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.105582Z","title":"Woodward","venue":null,"work_id":"03d48edb-cadf-4807-a581-dda5eacc8729","year":2003},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.081518Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b3e9c14be19040c69a2ec77972c42cc86c069a26e6cdba5a113b142a2a1d01cc","observation_id":"0c046019-3c7f-41e4-9991-f9a048c1a470","resolution":{"observed_at":"2026-08-16T04:25:07.111381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07476","last_updated":"2025-01-24T23:41:37Z","snapshot_observed_at":"2026-08-17T15:17:49.596382Z","submitted_at":"2024-10-09T23:02:00Z","title":"Towards a unified and verified understanding of group-operation networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07476","snapshot_observed_at":"2026-08-16T04:25:06.086534Z","title":"Unifying and verifying mechanistic interpretations: A case study with group operations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.086534Z"},"links":{"cited_paper":"/paper/2410.07476","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6fd2197c8a290ae4bbe31f8bea16d14e238759e8f049071f4e2267eec98fb3e9","observation_id":"3ff95413-4b01-4574-8af5-5fffe4e5c756","resolution":{"observed_at":"2026-08-16T04:25:06.086534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17148","last_updated":"2025-03-03T21:15:30Z","snapshot_observed_at":"2026-08-16T12:58:32.947185Z","submitted_at":"2025-01-28T18:51:24Z","title":"AxBench: Steering LLMs? Even Simple Baselines Outperform Sparse Autoencoders","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17148","snapshot_observed_at":"2026-08-16T04:25:06.093358Z","title":"Manning, and Christopher Potts","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.093358Z"},"links":{"cited_paper":"/paper/2501.17148","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:229447b604c462ed0981f2175a730b1e400b7ce151a06df6bea53c803e7f167c","observation_id":"74a24061-0e17-4599-a309-ec652df7de8d","resolution":{"observed_at":"2026-08-16T04:25:06.093358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.101369Z","title":"A theory of usable information under computational constraints","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.101369Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:fe210f7ec12635caf8ee5ce33c06e082b9a51517c9ca6154d43716eac624bb7c","observation_id":"5268333d-c642-4bbb-b2eb-f957abac149a","resolution":{"observed_at":"2026-08-16T04:25:06.101369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.070612Z","title":"Locally decodable codes","venue":null,"work_id":"44c36a21-d59d-4b9d-b0ca-7336a3b57e73","year":2012},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.108473Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:757b14cc76a03e8cbc38188930c447e6b4d3cc470a129bd7a31cc5f52b699a96","observation_id":"7474a404-cacd-4e6c-bd4e-726f49f34e19","resolution":{"observed_at":"2026-08-16T04:25:07.076567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T18:31:03.953403Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":62,"verified_exact":5,"verified_fuzzy":33},"total_outbound_references":105},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 100 of 105 outbound references and 1 inbound Pith citation observation for arXiv:2505.01372."}