{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BQF3XB6NRAPDGNT4VACNJWIWLA","short_pith_number":"pith:BQF3XB6N","schema_version":"1.0","canonical_sha256":"0c0bbb87cd881e33367ca804d4d916582dee8797d8fe0ce83723158e13cf94a0","source":{"kind":"arxiv","id":"2501.16615","version":2},"attestation_state":"computed","paper":{"title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gon\\c{c}alo Paulo, Nora Belrose","submitted_at":"2025-01-28T01:24:16Z","abstract_excerpt":"Sparse autoencoders (SAEs) are a useful tool for uncovering human-interpretable features in the activations of large language models (LLMs). While some expect SAEs to find the true underlying features used by a model, our research shows that SAEs trained on the same model and data, differing only in the random seed used to initialize their weights, identify different sets of features. For example, in an SAE with 131K latents trained on a feedforward network in Llama 3 8B, only 30% of the features were shared across different seeds. We observed this phenomenon across multiple layers of three di"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16615","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-28T01:24:16Z","cross_cats_sorted":[],"title_canon_sha256":"e10e36fcd5b5af996d3085a3c9060f7c4fbb1e02e3d8eb4c242d7f343e564c26","abstract_canon_sha256":"38079006b9df45a931db774508425daf59476ee24ede8c1b09c66a707a909188"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:02.939712Z","signature_b64":"wvs0vsn5wCjJF+QX2ds+rQPgEdNfz/xJM/Dh3tQPCpRx4n6s/TzphQVI1cQeCAOhMEQU4sed/wXULFU40PnSDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c0bbb87cd881e33367ca804d4d916582dee8797d8fe0ce83723158e13cf94a0","last_reissued_at":"2026-07-05T10:07:02.939249Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:02.939249Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparse Autoencoders Trained on the Same Data Learn Different Features","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Gon\\c{c}alo Paulo, Nora Belrose","submitted_at":"2025-01-28T01:24:16Z","abstract_excerpt":"Sparse autoencoders (SAEs) are a useful tool for uncovering human-interpretable features in the activations of large language models (LLMs). While some expect SAEs to find the true underlying features used by a model, our research shows that SAEs trained on the same model and data, differing only in the random seed used to initialize their weights, identify different sets of features. For example, in an SAE with 131K latents trained on a feedforward network in Llama 3 8B, only 30% of the features were shared across different seeds. We observed this phenomenon across multiple layers of three di"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16615","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16615/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16615","created_at":"2026-07-05T10:07:02.939305+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16615v2","created_at":"2026-07-05T10:07:02.939305+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16615","created_at":"2026-07-05T10:07:02.939305+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQF3XB6NRAPD","created_at":"2026-07-05T10:07:02.939305+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQF3XB6NRAPDGNT4","created_at":"2026-07-05T10:07:02.939305+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQF3XB6N","created_at":"2026-07-05T10:07:02.939305+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03002","citing_title":"Perplexity Can Miss SAE Feature Damage Under Quantization","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28149","citing_title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12138","citing_title":"Unstable Features, Reproducible Subspaces: Understanding Seed Dependence in Sparse Autoencoders","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26396","citing_title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06655","citing_title":"Graph-Regularized Sparse Autoencoders for LLM Safety Steering","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14347","citing_title":"Exemplar Partitioning for Mechanistic Interpretability","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18229","citing_title":"Are Sparse Autoencoder Benchmarks Reliable?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09708","citing_title":"Beyond I'm Sorry, I Can't: Dissecting Large Language Model Refusal","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14347","citing_title":"Exemplar Partitioning for Mechanistic Interpretability","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12874","citing_title":"Descriptive Collision in Sparse Autoencoder Auto-Interpretability: When One Explanation Describes Many Features","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04072","citing_title":"Sparse Autoencoder Decomposition of Clinical Sequence Model Representations: Feature Complexity, Task Specialisation, and Mortality Prediction","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA","json":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA.json","graph_json":"https://pith.science/api/pith-number/BQF3XB6NRAPDGNT4VACNJWIWLA/graph.json","events_json":"https://pith.science/api/pith-number/BQF3XB6NRAPDGNT4VACNJWIWLA/events.json","paper":"https://pith.science/paper/BQF3XB6N"},"agent_actions":{"view_html":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA","download_json":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA.json","view_paper":"https://pith.science/paper/BQF3XB6N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16615&json=true","fetch_graph":"https://pith.science/api/pith-number/BQF3XB6NRAPDGNT4VACNJWIWLA/graph.json","fetch_events":"https://pith.science/api/pith-number/BQF3XB6NRAPDGNT4VACNJWIWLA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA/action/storage_attestation","attest_author":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA/action/author_attestation","sign_citation":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA/action/citation_signature","submit_replication":"https://pith.science/pith/BQF3XB6NRAPDGNT4VACNJWIWLA/action/replication_record"}},"created_at":"2026-07-05T10:07:02.939305+00:00","updated_at":"2026-07-05T10:07:02.939305+00:00"}